{"manifest":{"schemaVersion":"1.0","dataAsOf":"2026-08-30","recordCount":2343,"classicRecordCount":39,"catalogSourceRecordCount":1094,"catalogEntityCount":910,"catalogMergedCount":57,"catalogOnlyCount":853,"recentRecordCount":1451,"scope":"all-time Library, complete BenchLM and llm-stats catalogs, plus recent Radar records"},"records":[{"id":"bm_360cityarena_2a62b93d","familyId":"bmf_28f4efc5217e","name":"360CityArena","oneLine":"360CityArena evaluates embodied agents in a photorealistic virtual urban environment built from 360-degree video of Tokyo's Akihabara district. It includes 175 tasks across environment understanding, path reasoning, and spatial reasoning, testing localization, landmark search, path planning, and relational spatial reasoning. Scoring uses mean evaluation score with varied success criteria.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.08814","pdf":"https://arxiv.org/pdf/2608.08814","project":"https://360mm-team.github.io/360CityArena/","code":"https://github.com/360MM-Team/360CityArena","data":null,"hfPaper":"https://huggingface.co/papers/2608.08814"},"evidence":{"snippet":"We present 360CityArena, a benchmark for evaluating the urban exploration capabilities of embodied agents within a photorealistic environment constructed from 360-degree videos.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":10,"hfDailySubmittedAt":"2026-08-12T00:00:00.000Z","githubStars":21,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08814"},"ranking":{"30d":{"score":46,"rank":40,"coverage":0.85,"confidence":"High"},"90d":{"score":45,"rank":114,"coverage":0.7,"confidence":"Medium"}},"description":"360CityArena evaluates embodied agents in a photorealistic virtual urban environment built from 360-degree video of Tokyo's Akihabara district. It includes 175 tasks across environment understanding, path reasoning, and spatial reasoning, testing localization, landmark search, path planning, and relational spatial reasoning. Scoring uses mean evaluation score with varied success criteria.","whyItMatters":"Existing outdoor navigation benchmarks lack photorealism or complexity, leaving a gap between simulated and real-world urban conditions. This benchmark provides a repeatable protocol with released code, tasks, and evaluation scripts, enabling comparative assessment of embodied agents' city-scale navigation and reasoning abilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2170a2e8e48844a0ccd97d4d848ea0c771da141cf2082f6ca914a5eab7d982f3"},"motivation":"We present 360CityArena, a benchmark for evaluating the urban exploration capabilities of embodied agents within a photorealistic environment constructed from 360-degree videos.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.08814","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"360MM Team","organizationType":"academic-lab","sourceUrl":"https://github.com/360MM-Team/360CityArena","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_3dcodebench_14ee1d5d","familyId":"bmf_151d2c52aca3","name":"3DCodeBench","oneLine":"3DCodeBench evaluates vision-language model agents on procedural 3D modeling by converting text and image references into Blender Python code. It includes 212 object categories with ground-truth scripts, and scores outputs on executability, image similarity (SigLIP-2/DINOv3), 3D shape distance (Chamfer/Uni3D), and LLM-as-judge, plus a human-preference ranking platform.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01057","pdf":"https://arxiv.org/pdf/2606.01057","project":"https://www.3dcodebench.com/","code":"https://github.com/gaoypeng/3dcodebench","data":null,"hfPaper":"https://huggingface.co/papers/2606.01057"},"evidence":{"snippet":"In this paper, we propose 3DCodeBench, a systematic benchmark for evaluating vision-language model (VLM) agents for procedural 3D generation in 3D modeling software.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":null,"githubStars":91,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01057"},"ranking":{},"description":"3DCodeBench evaluates vision-language model agents on procedural 3D modeling by converting text and image references into Blender Python code. It includes 212 object categories with ground-truth scripts, and scores outputs on executability, image similarity (SigLIP-2/DINOv3), 3D shape distance (Chamfer/Uni3D), and LLM-as-judge, plus a human-preference ranking platform.","whyItMatters":"There is no standardized way to compare model abilities in procedural 3D code generation. This benchmark provides a fixed protocol and dataset, enabling reproducible evaluation and comparison across models and coding-agent settings, useful for selecting models or guiding development of procedural modeling capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e46eeef8ddb34438972594cea22941243dba369ff169288b03aa926d38021f6a"},"motivation":"Procedural 3D modeling through code is emerging as a versatile paradigm, offering deterministic, engine-ready, and precisely editable assets that neural 3D generators inherently lack.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01057","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Google","organizationType":"company-research-lab","sourceUrl":"https://github.com/gaoypeng/3dcodebench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_4dsynth-controllable-procedural-world-synt_fa98bfb4","familyId":"bmf_fea82a91ffa7","name":"4DSynth-Nav","oneLine":"Evaluates embodied agents on interactive navigation tasks in procedurally generated 4D environments with independently tunable difficulty axes.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.26947","pdf":"https://arxiv.org/pdf/2608.26947","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"This paper presents both a controllable generation pipeline and the scalable benchmark it enables, offering a practical foundation for developing and evaluating embodied agents.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","named entity is a model, method, framework, or dataset used in evaluation"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":null,"hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26947"},"ranking":{},"description":"Evaluates embodied agents on interactive navigation tasks in procedurally generated 4D environments with independently tunable difficulty axes.","whyItMatters":"Offers a scalable, controllable benchmark for embodied navigation that enables reproducible failure analysis and difficulty modulation.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T07:23:36.749451Z","inputHash":"92c5e8fabb3d5ac8ae2535731e816fc7cb1dd57eff6e11ec60e1d68abf89b6e4"},"motivation":"Embodied agents need environments that are visually diverse, physically interactive, and changing over time.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T07:23:36.749451Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is part of the 4DSynth system release, which includes procedural generation and the benchmark itself, providing a public reuse path.","canonicalNameSource":"abstract","canonicalNameEvidence":"we construct 4DSynth-Nav, an interactive navigation benchmark generated entirely from 4DSynth's procedural scenes."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.26947","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-30T06:59:55.242506Z"},"attentionForecast":{"score":40,"confidence":"Low","horizon":"7d","reason":"The focus on embodied simulation and navigation may attract a specialized audience, but the procedural controllability adds novelty that could drive interest."},"evaluationMode":"public_reusable","publishers":[{"name":"4DSynth authors","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.26947","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_a-large-scale-dataset-and-benchmark_80e45544","familyId":"bmf_06e6d8e2fe19","name":"A Large-Scale Dataset and Benchmark","oneLine":"InteractBind provides a large-scale dataset of ~100k protein-ligand pairs with fine-grained binding-site localization tasks and interaction maps for six non-covalent interaction types, plus affinity and similarity-controlled splits.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24045","pdf":"https://arxiv.org/pdf/2605.24045","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24045"},"evidence":{"snippet":"To address this gap, we introduce InteractBind, a large-scale protein-ligand dataset comprising approximately 100k protein-ligand pairs, together with a benchmark for fine-grained evaluation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24045"},"ranking":{},"description":"InteractBind provides a large-scale dataset of ~100k protein-ligand pairs with fine-grained binding-site localization tasks and interaction maps for six non-covalent interaction types, plus affinity and similarity-controlled splits.","whyItMatters":"Challenges existing protein-ligand benchmarks by focusing on localization and interaction interpretability, which is critical for drug discovery.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d9fc988423d4f52a06e3da234630e597ddf74eca1f6a54f647a6ba78aa296eb2"},"motivation":"Protein-ligand modeling underpins computational drug discovery and molecular design.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24045","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_f5f45072d29f09b9","familyId":"catalog_family_f5f45072d29f09b9","name":"AA Agentic Index","oneLine":"A display-only Artificial Analysis agentic index.","description":"A display-only Artificial Analysis agentic index.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/leaderboards/models","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f5f45072d29f09b9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aaagenticindex"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaAgenticIndex","url":"https://benchlm.ai/benchmarks/aaagenticindex","paperUrl":"https://artificialanalysis.ai/leaderboards/models","year":"2026","fullName":"Artificial Analysis Agentic Index","format":"Aggregated model score","tasks":"Cross-benchmark agentic index","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_38b40f0190dcb2ed","familyId":"catalog_family_38b40f0190dcb2ed","name":"AA AIME 2025","oneLine":"An independently evaluated AIME 2025 result from Artificial Analysis.","description":"An independently evaluated AIME 2025 result from Artificial Analysis.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/aime-2025","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_38b40f0190dcb2ed"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aaaime2025"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaAime2025","url":"https://benchlm.ai/benchmarks/aaaime2025","paperUrl":"https://artificialanalysis.ai/evaluations/aime-2025","year":"2026","fullName":"Artificial Analysis AIME 2025","format":"Accuracy","tasks":"30 AIME 2025 problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_a28b9404bc9681d9","familyId":"catalog_family_a28b9404bc9681d9","name":"AA AutomationBench","oneLine":"An independently evaluated automation benchmark from Artificial Analysis.","description":"An independently evaluated automation benchmark from Artificial Analysis.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/automationbench-aa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a28b9404bc9681d9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aaautomationbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaAutomationBench","url":"https://benchlm.ai/benchmarks/aaautomationbench","paperUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa","year":"2026","fullName":"Artificial Analysis AutomationBench","format":"Task success rate","tasks":"Business-process automation tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_197d4185398ce127","familyId":"catalog_family_197d4185398ce127","name":"AA Briefcase","oneLine":"AA-Briefcase is an Artificial Analysis evaluation of AI systems on professional knowledge-work tasks, reported as an Elo score.","description":"AA-Briefcase is an Artificial Analysis evaluation of AI systems on professional knowledge-work tasks, reported as an Elo score.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Productivity","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/aa-briefcase","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_197d4185398ce127"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aabriefcaseelo"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/aa-briefcase"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaBriefcaseElo","url":"https://benchlm.ai/benchmarks/aabriefcaseelo","paperUrl":"https://artificialanalysis.ai/evaluations/aa-briefcase","year":"2026","fullName":"Artificial Analysis Briefcase","format":"Elo","tasks":"Professional knowledge-work tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"aa-briefcase","url":"https://llm-stats.com/benchmarks/aa-briefcase","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","productivity","reasoning","agents"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_e5e2e35ac6eefdb9","familyId":"catalog_family_e5e2e35ac6eefdb9","name":"AA Coding Agents","oneLine":"A display-only Artificial Analysis leaderboard for coding-agent systems, combining agent harnesses, host models, and execution settings across software-engineering benchmarks.","description":"A display-only Artificial Analysis leaderboard for coding-agent systems, combining agent harnesses, host models, and execution settings across software-engineering benchmarks.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/agents/coding-agents","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e5e2e35ac6eefdb9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aacodingagents"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaCodingAgents","url":"https://benchlm.ai/benchmarks/aacodingagents","paperUrl":"https://artificialanalysis.ai/agents/coding-agents","year":"2026","fullName":"Artificial Analysis Coding Agent Index","format":"Average pass@1 index","tasks":"Composite over DeepSWE, Terminal-Bench v2, and SWE-Atlas-QnA","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_81316bf8cb0c02b7","familyId":"catalog_family_81316bf8cb0c02b7","name":"AA Coding Index","oneLine":"A display-only Artificial Analysis coding index.","description":"A display-only Artificial Analysis coding index.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/leaderboards/models","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_81316bf8cb0c02b7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aacodingindex"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaCodingIndex","url":"https://benchlm.ai/benchmarks/aacodingindex","paperUrl":"https://artificialanalysis.ai/leaderboards/models","year":"2026","fullName":"Artificial Analysis Coding Index","format":"Aggregated model score","tasks":"Cross-benchmark coding index","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_8e4a8ec02cfd3740","familyId":"catalog_family_8e4a8ec02cfd3740","name":"AA EnterpriseOps-Gym","oneLine":"An independently evaluated enterprise-operations benchmark from Artificial Analysis.","description":"An independently evaluated enterprise-operations benchmark from Artificial Analysis.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8e4a8ec02cfd3740"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aaenterpriseopsgym"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaEnterpriseOpsGym","url":"https://benchlm.ai/benchmarks/aaenterpriseopsgym","paperUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","year":"2026","fullName":"Artificial Analysis EnterpriseOps-Gym","format":"Task success rate","tasks":"Enterprise operations workflows","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_5d15918fed136b3a","familyId":"catalog_family_5d15918fed136b3a","name":"AA Global-MMLU-Lite","oneLine":"An independently evaluated multilingual knowledge result from Artificial Analysis.","description":"An independently evaluated multilingual knowledge result from Artificial Analysis.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multilingual"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/global-mmlu-lite","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5d15918fed136b3a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aaglobalmmlulite"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaGlobalMmluLite","url":"https://benchlm.ai/benchmarks/aaglobalmmlulite","paperUrl":"https://artificialanalysis.ai/evaluations/global-mmlu-lite","year":"2026","fullName":"Artificial Analysis Global-MMLU-Lite","format":"Accuracy","tasks":"Multilingual knowledge questions","successorKey":null}],"catalogCategories":["multilingual"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d1a0b521e153e76c","familyId":"catalog_family_d1a0b521e153e76c","name":"AA Harvey LAB","oneLine":"An independently evaluated legal-agent benchmark from Artificial Analysis.","description":"An independently evaluated legal-agent benchmark from Artificial Analysis.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d1a0b521e153e76c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aaharveylab"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaHarveyLab","url":"https://benchlm.ai/benchmarks/aaharveylab","paperUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","year":"2026","fullName":"Artificial Analysis Harvey LAB-AA","format":"Task success rate","tasks":"Legal agent tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_39e22f8f4a542156","familyId":"catalog_family_39e22f8f4a542156","name":"AA ITBench","oneLine":"An independently evaluated IT-operations benchmark from Artificial Analysis.","description":"An independently evaluated IT-operations benchmark from Artificial Analysis.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/itbench-aa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_39e22f8f4a542156"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aaitbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaItbench","url":"https://benchlm.ai/benchmarks/aaitbench","paperUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","year":"2026","fullName":"Artificial Analysis ITBench-AA","format":"Task success rate","tasks":"IT incident-response tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d2d201bf0ef79343","familyId":"catalog_family_d2d201bf0ef79343","name":"AA LiveCodeBench","oneLine":"An independently evaluated LiveCodeBench result from Artificial Analysis.","description":"An independently evaluated LiveCodeBench result from Artificial Analysis.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/livecodebench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d2d201bf0ef79343"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aalivecodebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaLiveCodeBench","url":"https://benchlm.ai/benchmarks/aalivecodebench","paperUrl":"https://artificialanalysis.ai/evaluations/livecodebench","year":"2026","fullName":"Artificial Analysis LiveCodeBench","format":"Pass rate","tasks":"Contamination-resistant coding tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_ec021b89ae0af531","familyId":"catalog_family_ec021b89ae0af531","name":"AA MATH-500","oneLine":"An independently evaluated MATH-500 result from Artificial Analysis.","description":"An independently evaluated MATH-500 result from Artificial Analysis.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/math-500","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ec021b89ae0af531"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aamath500"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaMath500","url":"https://benchlm.ai/benchmarks/aamath500","paperUrl":"https://artificialanalysis.ai/evaluations/math-500","year":"2026","fullName":"Artificial Analysis MATH-500","format":"Accuracy","tasks":"500 competition mathematics problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_10defe438c87d701","familyId":"catalog_family_10defe438c87d701","name":"AA MMLU-Pro","oneLine":"An independently evaluated MMLU-Pro result from Artificial Analysis.","description":"An independently evaluated MMLU-Pro result from Artificial Analysis.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/mmlu-pro","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_10defe438c87d701"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aammlupro"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaMmluPro","url":"https://benchlm.ai/benchmarks/aammlupro","paperUrl":"https://artificialanalysis.ai/evaluations/mmlu-pro","year":"2026","fullName":"Artificial Analysis MMLU-Pro","format":"Accuracy","tasks":"Professional multi-subject questions","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_19f62069d90ec4ef","familyId":"catalog_family_19f62069d90ec4ef","name":"AA Openness Index","oneLine":"A display-only Artificial Analysis model-openness index.","description":"A display-only Artificial Analysis model-openness index.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/artificial-analysis-openness-index","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_19f62069d90ec4ef"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aaopennessindex"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaOpennessIndex","url":"https://benchlm.ai/benchmarks/aaopennessindex","paperUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-openness-index","year":"2026","fullName":"Artificial Analysis Openness Index","format":"Index score","tasks":"Model openness assessment","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e1c010541a3a1310","familyId":"catalog_family_e1c010541a3a1310","name":"AA Tau3 Banking","oneLine":"An independently evaluated Tau3 banking benchmark from Artificial Analysis.","description":"An independently evaluated Tau3 banking benchmark from Artificial Analysis.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/tau3-banking","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e1c010541a3a1310"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aatau3banking"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaTau3Banking","url":"https://benchlm.ai/benchmarks/aatau3banking","paperUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","year":"2026","fullName":"Artificial Analysis Tau3-Banking","format":"Task success rate","tasks":"Banking tool-use workflows","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0218154871e09c0f","familyId":"catalog_family_0218154871e09c0f","name":"AA Terminal-Bench 2.1","oneLine":"An independently evaluated Terminal-Bench v2.1 result from Artificial Analysis.","description":"An independently evaluated Terminal-Bench v2.1 result from Artificial Analysis.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/terminalbench-v2-1","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0218154871e09c0f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aaterminalbench21"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaTerminalBench21","url":"https://benchlm.ai/benchmarks/aaterminalbench21","paperUrl":"https://artificialanalysis.ai/evaluations/terminalbench-v2-1","year":"2026","fullName":"Artificial Analysis Terminal-Bench v2.1","format":"Task success rate","tasks":"Terminal-based agent tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_c02c7aee95c20159","familyId":"catalog_family_c02c7aee95c20159","name":"AA-GPQA Diamond","oneLine":"A display-only Artificial Analysis GPQA Diamond score.","description":"A display-only Artificial Analysis GPQA Diamond score.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/gpqa-diamond","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c02c7aee95c20159"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aagpqadiamond"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaGpqaDiamond","url":"https://benchlm.ai/benchmarks/aagpqadiamond","paperUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","year":"2026","fullName":"Artificial Analysis GPQA Diamond","format":"Accuracy","tasks":"Graduate-level science questions","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_6e6434e20772e9a7","familyId":"catalog_family_6e6434e20772e9a7","name":"AA-HLE","oneLine":"A display-only Artificial Analysis Humanity's Last Exam score.","description":"A display-only Artificial Analysis Humanity's Last Exam score.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/hle","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6e6434e20772e9a7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aahle"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaHle","url":"https://benchlm.ai/benchmarks/aahle","paperUrl":"https://artificialanalysis.ai/evaluations/hle","year":"2026","fullName":"Artificial Analysis Humanity's Last Exam","format":"Accuracy","tasks":"Expert-level questions","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_3ab41c34d802c79d","familyId":"catalog_family_3ab41c34d802c79d","name":"AA-IFBench","oneLine":"A display-only Artificial Analysis IFBench score.","description":"A display-only Artificial Analysis IFBench score.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Instructionfollowing"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/ifbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3ab41c34d802c79d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aaifbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaIfBench","url":"https://benchlm.ai/benchmarks/aaifbench","paperUrl":"https://artificialanalysis.ai/evaluations/ifbench","year":"2026","fullName":"Artificial Analysis IFBench","format":"Constraint satisfaction accuracy","tasks":"Verifiable instruction constraints","successorKey":null}],"catalogCategories":["instructionFollowing"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_c6adbc64ef2ca87b","familyId":"catalog_family_c6adbc64ef2ca87b","name":"AA-Index","oneLine":"No official academic documentation found for this benchmark. Extensive research through ArXiv, IEEE/ACL/NeurIPS papers, and university research sites yielded no peer-reviewed sources for an 'aa-index' benchmark. This entry requires verification from official academic sources.","description":"No official academic documentation found for this benchmark. Extensive research through ArXiv, IEEE/ACL/NeurIPS papers, and university research sites yielded no peer-reviewed sources for an 'aa-index' benchmark. This entry requires verification from official academic sources.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/aa-index","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c6adbc64ef2ca87b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/aa-index"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"aa-index","url":"https://llm-stats.com/benchmarks/aa-index","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["general"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_c847be0a0dc9793c","familyId":"catalog_family_c847be0a0dc9793c","name":"AA-LCR","oneLine":"Agent Arena Long Context Reasoning benchmark","description":"Agent Arena Long Context Reasoning benchmark","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/models/grok-4-3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c847be0a0dc9793c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/lcr"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/aa-lcr"}],"catalogSources":[{"catalog":"benchlm","sourceId":"lcr","url":"https://benchlm.ai/benchmarks/lcr","paperUrl":"https://artificialanalysis.ai/models/grok-4-3","year":"2026","fullName":"Artificial Analysis Long Context Reasoning","format":"Accuracy","tasks":"Long-context reasoning tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"aa-lcr","url":"https://llm-stats.com/benchmarks/aa-lcr","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","long context"],"catalogModelCount":19,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_543b3e1b707ea114","familyId":"catalog_family_543b3e1b707ea114","name":"AA-MMMU-Pro","oneLine":"A display-only Artificial Analysis MMMU-Pro score.","description":"A display-only Artificial Analysis MMMU-Pro score.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/mmmu-pro","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_543b3e1b707ea114"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aammmupro"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaMmmuPro","url":"https://benchlm.ai/benchmarks/aammmupro","paperUrl":"https://artificialanalysis.ai/evaluations/mmmu-pro","year":"2026","fullName":"Artificial Analysis MMMU-Pro","format":"Image + text question answering","tasks":"Multimodal academic reasoning","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_208ea5652cf3cd61","familyId":"catalog_family_208ea5652cf3cd61","name":"AA-Omniscience Accuracy","oneLine":"A display-only Artificial Analysis knowledge metric for the proportion of correctly answered questions.","description":"A display-only Artificial Analysis knowledge metric for the proportion of correctly answered questions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/models/grok-4-3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_208ea5652cf3cd61"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/omniscienceaccuracy"}],"catalogSources":[{"catalog":"benchlm","sourceId":"omniscienceAccuracy","url":"https://benchlm.ai/benchmarks/omniscienceaccuracy","paperUrl":"https://artificialanalysis.ai/models/grok-4-3","year":"2026","fullName":"Artificial Analysis Omniscience Accuracy","format":"Accuracy","tasks":"Knowledge questions","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_16d2024b3229447d","familyId":"catalog_family_16d2024b3229447d","name":"AA-Omniscience Hallucination Rate","oneLine":"A display-only Artificial Analysis factuality metric for the rate of incorrect answers among non-correct responses.","description":"A display-only Artificial Analysis factuality metric for the rate of incorrect answers among non-correct responses.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/models/grok-4-3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_16d2024b3229447d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/omnisciencehallucinationrate"}],"catalogSources":[{"catalog":"benchlm","sourceId":"omniscienceHallucinationRate","url":"https://benchlm.ai/benchmarks/omnisciencehallucinationrate","paperUrl":"https://artificialanalysis.ai/models/grok-4-3","year":"2026","fullName":"Artificial Analysis Omniscience Hallucination Rate","format":"Hallucination rate","tasks":"Knowledge questions","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_4bfe2ff9ab36de99","familyId":"catalog_family_4bfe2ff9ab36de99","name":"AA-Omniscience Index","oneLine":"AA-Omniscience Index is Artificial Analysis's knowledge-reliability metric. It rewards correct answers, penalizes hallucinations, and does not penalize abstention. Scores range from -100 to 100, where 0 means as many correct as incorrect answers.","description":"AA-Omniscience Index is Artificial Analysis's knowledge-reliability metric. It rewards correct answers, penalizes hallucinations, and does not penalize abstention. Scores range from -100 to 100, where 0 means as many correct as incorrect answers.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","Science"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/omniscience","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4bfe2ff9ab36de99"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aaomniscienceindex"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/aa-omniscience-index"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaOmniscienceIndex","url":"https://benchlm.ai/benchmarks/aaomniscienceindex","paperUrl":"https://artificialanalysis.ai/evaluations/omniscience","year":"2026","fullName":"Artificial Analysis Omniscience Index","format":"Index score","tasks":"Knowledge questions","successorKey":null},{"catalog":"llm-stats","sourceId":"aa-omniscience-index","url":"https://llm-stats.com/benchmarks/aa-omniscience-index","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","science"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_6fb1a12a623ab41b","familyId":"catalog_family_6fb1a12a623ab41b","name":"AA-SciCode","oneLine":"A display-only Artificial Analysis SciCode score.","description":"A display-only Artificial Analysis SciCode score.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/scicode","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6fb1a12a623ab41b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aascicode"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aaSciCode","url":"https://benchlm.ai/benchmarks/aascicode","paperUrl":"https://artificialanalysis.ai/evaluations/scicode","year":"2026","fullName":"Artificial Analysis SciCode","format":"Task success rate","tasks":"Scientific coding subproblems","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_aarri-bench_d4d9e662","familyId":"bmf_fc38d9b8cce3","name":"AARRI-Bench","oneLine":"AARRI-Bench evaluates LLM agents on entry-level research intern tasks, measuring success rate in containerized environments with fixed tasks and scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07462","pdf":"https://arxiv.org/pdf/2606.07462","project":null,"code":"https://github.com/AARR-bench/AARRI-bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.07462"},"evidence":{"snippet":"In this work, we propose AARRI-Bench (Act As a Real Research Intern), the first benchmark in this series.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":11,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07462"},"ranking":{"90d":{"score":41,"rank":142,"coverage":0.55,"confidence":"Low"}},"description":"AARRI-Bench evaluates LLM agents on entry-level research intern tasks, measuring success rate in containerized environments with fixed tasks and scoring.","whyItMatters":"Provides a reproducible benchmark for agentic research behavior, highlighting gaps in nuanced reasoning and offering a public comparison platform.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3db4d6d91a57e40f23b85b7e3721dd78da7795db33d90eec351081dc5a90e493"},"motivation":"As foundation models advance and agent scaffolding becomes increasingly sophisticated, agents have demonstrated remarkable proficiency in complex, long-horizon coding tasks and even autonomous experiment execution.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07462","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AARR-bench","organizationType":"benchmark-organization","sourceUrl":"https://github.com/AARR-bench/AARRI-bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_abc-bench_fecd3eb4","familyId":"bmf_5fb363ebd5d3","name":"ABC-Bench","oneLine":"ABC-Bench evaluates LLM agents on biosecurity-relevant tasks including liquid handling robot code generation, DNA fragment design, and DNA synthesis screening evasion, with wet-lab validation.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences","Robotics & Autonomous Systems"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech","Robotics"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11150","pdf":"https://arxiv.org/pdf/2606.11150","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.11150"},"evidence":{"snippet":"To address this, we introduce the Agentic Bio-Capabilities Benchmark (ABC-Bench), a suite of tasks to measure agentic biosecurity-relevant capabilities.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.11150"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ABC-Bench evaluates LLM agents on biosecurity-relevant tasks including liquid handling robot code generation, DNA fragment design, and DNA synthesis screening evasion, with wet-lab validation.","whyItMatters":"Measures agentic AI capabilities relevant to biosecurity, offering a standardized protocol to assess dual-use risks and inform safeguards in biological research contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"46382a337c2a4c2787994e7a9a1c935baafe0b8dcfffb05409ef80c2b771a938"},"motivation":"Large language models (LLMs) are rapidly acquiring capabilities relevant to biological research, from literature synthesis to interpretation of experimental data.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11150","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"catalog_500ebaeab92cc8d8","familyId":"catalog_family_500ebaeab92cc8d8","name":"ACCR Daybreak Blue","oneLine":"Share of approved advanced-cyber requests completed rather than refused by GPT-5.6 Sol under Daybreak Blue safeguards.","description":"Share of approved advanced-cyber requests completed rather than refused by GPT-5.6 Sol under Daybreak Blue safeguards.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/expanding-daybreak-as-the-cyber-defense-window-narrows/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_500ebaeab92cc8d8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/advancedcybercompletionrateblue"}],"catalogSources":[{"catalog":"benchlm","sourceId":"advancedCyberCompletionRateBlue","url":"https://benchlm.ai/benchmarks/advancedcybercompletionrateblue","paperUrl":"https://openai.com/index/expanding-daybreak-as-the-cyber-defense-window-narrows/","year":"2026","fullName":"Advanced Cyber Completion Rate — Daybreak Blue","format":"Completion rate","tasks":"Internal advanced-cyber request set","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_08196beb56175bcd","familyId":"catalog_family_08196beb56175bcd","name":"ACCR Daybreak Red","oneLine":"Share of approved advanced-cyber requests completed rather than refused by GPT-5.6 Cyber under Daybreak Red access.","description":"Share of approved advanced-cyber requests completed rather than refused by GPT-5.6 Cyber under Daybreak Red access.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/expanding-daybreak-as-the-cyber-defense-window-narrows/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_08196beb56175bcd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/advancedcybercompletionratered"}],"catalogSources":[{"catalog":"benchlm","sourceId":"advancedCyberCompletionRateRed","url":"https://benchlm.ai/benchmarks/advancedcybercompletionratered","paperUrl":"https://openai.com/index/expanding-daybreak-as-the-cyber-defense-window-narrows/","year":"2026","fullName":"Advanced Cyber Completion Rate — Daybreak Red","format":"Completion rate","tasks":"Internal advanced-cyber request set","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_f37de3e9602e8bbc","familyId":"catalog_family_f37de3e9602e8bbc","name":"ACCR standard","oneLine":"Share of approved advanced-cyber requests completed rather than refused under standard GPT-5.6 Sol safeguards.","description":"Share of approved advanced-cyber requests completed rather than refused under standard GPT-5.6 Sol safeguards.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/expanding-daybreak-as-the-cyber-defense-window-narrows/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f37de3e9602e8bbc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/advancedcybercompletionratestandard"}],"catalogSources":[{"catalog":"benchlm","sourceId":"advancedCyberCompletionRateStandard","url":"https://benchlm.ai/benchmarks/advancedcybercompletionratestandard","paperUrl":"https://openai.com/index/expanding-daybreak-as-the-cyber-defense-window-narrows/","year":"2026","fullName":"Advanced Cyber Completion Rate — Standard Access","format":"Completion rate","tasks":"Internal advanced-cyber request set","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_844a408311fdb1ec","familyId":"catalog_family_844a408311fdb1ec","name":"ACE solved","oneLine":"Number of advanced cyber-range challenges solved in the joint NIST CAISI and UK AISI preliminary evaluation.","description":"Number of advanced cyber-range challenges solved in the joint NIST CAISI and UK AISI preliminary evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_844a408311fdb1ec"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/acecyberrangesolved"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aceCyberRangeSolved","url":"https://benchlm.ai/benchmarks/acecyberrangesolved","paperUrl":"https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities","year":"2026","fullName":"ACE Cyber Range Challenges Solved","format":"Challenges solved","tasks":"41 advanced cyber-range challenges","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0a7b5bc3b63e66b3","familyId":"catalog_family_0a7b5bc3b63e66b3","name":"ACEBench","oneLine":"ACEBench is a comprehensive benchmark for evaluating Large Language Models' tool usage capabilities across three primary evaluation types: Normal (basic tool usage scenarios), Special (tool usage with ambiguous or incomplete instructions), and Agent (multi-agent interactions simulating real-world dialogues). The benchmark covers 4,538 APIs across 8 major domains and 68 sub-domains including technology, finance, entertainment, society, health, culture, and environment, supporting both English and Chinese languages.","description":"ACEBench is a comprehensive benchmark for evaluating Large Language Models' tool usage capabilities across three primary evaluation types: Normal (basic tool usage scenarios), Special (tool usage with ambiguous or incomplete instructions), and Agent (multi-agent interactions simulating real-world dialogues). The benchmark covers 4,538 APIs across 8 major domains and 68 sub-domains including technology, finance, entertainment, society, health, culture, and environment, supporting both English and Chinese languages.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Finance","General","Healthcare","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/acebench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0a7b5bc3b63e66b3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/acebench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"acebench","url":"https://llm-stats.com/benchmarks/acebench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","finance","general","healthcare","tool calling"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"specific"},{"id":"bm_active-swe_8c4ac0e6","familyId":"bmf_2cbe3d83caa7","name":"Active-SWE","oneLine":"Active-SWE evaluates coding agents on proactive bug fixing: detecting and fixing multiple bugs without issue reports. It includes 1,663 tasks across six bug categories and eight languages, with stages for recorded bugs, potential bugs, and judge validation.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04682","pdf":"https://arxiv.org/pdf/2608.04682","project":null,"code":"https://github.com/XLearning-SCU/Active-SWE","data":null,"hfPaper":"https://huggingface.co/papers/2608.04682"},"evidence":{"snippet":"To address this, we introduce Active-SWE, a benchmark for evaluating coding agents on proactively discovering and fixing multiple bugs without report guidance, covering 1,663 tasks across six bug categories and eight languages.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":4,"hfDailySubmittedAt":null,"githubStars":81,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04682"},"ranking":{"30d":{"score":52,"rank":21,"coverage":0.85,"confidence":"High"},"90d":{"score":52,"rank":52,"coverage":0.7,"confidence":"Medium"}},"description":"Active-SWE evaluates coding agents on proactive bug fixing: detecting and fixing multiple bugs without issue reports. It includes 1,663 tasks across six bug categories and eight languages, with stages for recorded bugs, potential bugs, and judge validation.","whyItMatters":"Existing SWE benchmarks assume detailed issue reports are available, which is unrealistic. Active-SWE fills the gap by testing agents' ability to discover and fix bugs proactively, a capability that current state-of-the-art agents struggle with, providing a more realistic evaluation of coding agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a7408611b0f4eb01effd4883f84e64de9af4d4454e81497eff0d21095bb55c7b"},"motivation":"Coding agents powered by large language models (LLMs) are increasingly adopted in software engineering (SWE) scenarios, capable of fixing a specific bug in large-scale codebase.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04682","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"XLearning-SCU","organizationType":"academic-lab","sourceUrl":"https://github.com/XLearning-SCU/Active-SWE","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_activefly-bench_205c2488","familyId":"bmf_c957b3cef25e","name":"ActiveFly-Bench","oneLine":"ActiveFly-Bench is a benchmark for UAV embodied perception, decomposing active perception into three tasks: Aerial Embodied Question Answering (Air-EQA), Observation Behavior Planning (OBP), and Fine-grained Language-guided UAV Control (FLUC). It includes datasets from real-world and simulated outdoor environments.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.10180","pdf":"https://arxiv.org/pdf/2607.10180","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.10180"},"evidence":{"snippet":"We introduce ActiveFly-Bench, the first benchmark to bridge cyberspace reasoning and physical-world interaction for UAV embodied perception.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.10180"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ActiveFly-Bench is a benchmark for UAV embodied perception, decomposing active perception into three tasks: Aerial Embodied Question Answering (Air-EQA), Observation Behavior Planning (OBP), and Fine-grained Language-guided UAV Control (FLUC). It includes datasets from real-world and simulated outdoor environments.","whyItMatters":"Active perception in UAVs requires bridging high-level reasoning with low-level control, which current VLMs and VLA models struggle with. This benchmark provides a testbed for embodied aerial intelligence, addressing the lack of benchmarks in this domain.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1bff2991bf2edec25de36a25b1cac71779617d5eab15572fb65c357a7605d9df"},"motivation":"We introduce ActiveFly-Bench, the first benchmark to bridge cyberspace reasoning and physical-world interaction for UAV embodied perception.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.10180","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_b152800e6ac82bd3","familyId":"catalog_family_b152800e6ac82bd3","name":"ActivityNet","oneLine":"A large-scale video benchmark for human activity understanding. Provides samples from 203 activity classes with an average of 137 untrimmed videos per class and 1.41 activity instances per video, for a total of 849 video hours. The benchmark covers a wide range of complex human activities that are of interest to people in their daily living and can be used to compare algorithms for three scenarios: untrimmed video classification, trimmed activity classification, and activity detection.","description":"A large-scale video benchmark for human activity understanding. Provides samples from 203 activity classes with an average of 137 untrimmed videos per class and 1.41 activity instances per video, for a total of 849 video hours. The benchmark covers a wide range of complex human activities that are of interest to people in their daily living and can be used to compare algorithms for three scenarios: untrimmed video classification, trimmed activity classification, and activity detection.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/activitynet","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b152800e6ac82bd3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/activitynet"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"activitynet","url":"https://llm-stats.com/benchmarks/activitynet","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["video","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_adamem-bench_de779711","familyId":"bmf_898bac33904d","name":"AdaMem-Bench","oneLine":"AdaMem-Bench simulates weeks of interaction with week-by-week question answering to evaluate memory policies for personalized long-horizon LLM agents. It measures QA accuracy and memory volume across different models.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-19","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.21144","pdf":"https://arxiv.org/pdf/2606.21144","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.21144"},"evidence":{"snippet":"To study this setting, we build \\textbf{AdaMem-Bench}, a benchmark that simulates weeks of interaction with week-by-week QA.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.21144"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AdaMem-Bench simulates weeks of interaction with week-by-week question answering to evaluate memory policies for personalized long-horizon LLM agents. It measures QA accuracy and memory volume across different models.","whyItMatters":"Long-term memory systems often bloat with irrelevant data. AdaMem-Bench provides a controlled environment to assess memory selection strategies, impacting efficiency and accuracy in personalized agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"292622571221fc7cc2b80aa551fadb6ec2333f2958bd85dec4230836e20d6e1c"},"motivation":"Long-term memory systems for Large Language Model (LLM) agents typically try to \\emph{remember everything}, extracting memories uniformly to retain as many facts as possible.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.21144","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AdaMem Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.21144","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_adaplanbench_b501c6c7","familyId":"bmf_72cbae1d4e60","name":"AdaPlanBench","oneLine":"Evaluates adaptive planning of LLM agents under progressively disclosed world and user constraints across 307 household tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05622","pdf":"https://arxiv.org/pdf/2606.05622","project":null,"code":"https://github.com/JiayuJeff/AdaPlanBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.05622"},"evidence":{"snippet":"To address this gap, we introduce AdaPlanBench, a dynamic interactive benchmark for evaluating whether Large Language Model (LLM) agents can adaptively plan and re-plan under progressively revealed world and user constraints.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":44,"hfDailySubmittedAt":"2026-06-05T00:00:00.000Z","githubStars":28,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05622"},"ranking":{"90d":{"score":50,"rank":63,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates adaptive planning of LLM agents under progressively disclosed world and user constraints across 307 household tasks.","whyItMatters":"Fills the gap in evaluating re-planning under dual constraints with interactive feedback, offering a testbed for reliable adaptation in LLM agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2e218ca1d00c479963f36a7e5815799bc769e2a1b6eeb17c35fb924c4f751e42"},"motivation":"Planning for real-world problems by language models often involves both world and user constraints, which may not be fully specified upfront and are progressively disclosed through interaction.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05622","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"JiayuJeff/AdaPlanBench","organizationType":"community","sourceUrl":"https://github.com/JiayuJeff/AdaPlanBench","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_adepts-bench_6cc5b43a","familyId":"bmf_c0573fbf9739","name":"ADeptS-Bench","oneLine":"Computer Use Agents (CUAs) are increasingly deployed to navigate mobile and desktop applications on behalf of users, yet no benchmark comprehensively evaluates whether they can safely interact with visual interfaces while handling ambiguou…","area":"Agents & Tool Use","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Manufacturing"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-28","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.26204","pdf":"https://arxiv.org/pdf/2608.26204","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.26204"},"evidence":{"snippet":"We introduce ADeptS-Bench, a dual-stream trustworthiness benchmark, grounded in the ADEPTS capability framework and general population user studies.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26204"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"Computer Use Agents (CUAs) are increasingly deployed to navigate mobile and desktop applications on behalf of users, yet no benchmark comprehensively evaluates whether they can safely interact with visual interfaces while handling ambiguous instructions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.26204","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_53b789eb0f0ba6dc","familyId":"catalog_family_53b789eb0f0ba6dc","name":"AdvancedIF","oneLine":"AdvancedIF is a rubric-based benchmark measuring complex, multi-turn, and system-prompted instruction following ability, scored with a calibrated LLM judge against per-instruction rubrics.","description":"AdvancedIF is a rubric-based benchmark measuring complex, multi-turn, and system-prompted instruction following ability, scored with a calibrated LLM judge against per-instruction rubrics.","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Instruction Following","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/advancedif","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_53b789eb0f0ba6dc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/advancedif"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"advancedif","url":"https://llm-stats.com/benchmarks/advancedif","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["instruction following","reasoning","general"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general"},{"id":"bm_advancedmathbench_22d95486","familyId":"bmf_aa5a0a0d601b","name":"AdvancedMathBench","oneLine":"AdvancedMathBench is a benchmark suite for advanced mathematical reasoning, containing ProverBench (296 proof problems) and VerifierBench (888 proof trajectories with expert labels), with an automatic verification pipeline.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.11849","pdf":"https://arxiv.org/pdf/2607.11849","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.11849"},"evidence":{"snippet":"To bridge this gap, we introduce AdvancedMathBench, a benchmark suite designed to evaluate advanced mathematical reasoning capabilities.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":33,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.11849"},"ranking":{"90d":{"score":52,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"AdvancedMathBench is a benchmark suite for advanced mathematical reasoning, containing ProverBench (296 proof problems) and VerifierBench (888 proof trajectories with expert labels), with an automatic verification pipeline.","whyItMatters":"Existing math benchmarks focus on high-school levels and final answers. AdvancedMathBench evaluates proof generation and verification at advanced levels, providing fine-grained assessments of proof correctness and error detection.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cf8c4038b5cb13c25589b51cafb5222643936f34d4177439d46d3b8dfc07f355"},"motivation":"Large language models (LLMs) have achieved remarkable performance on high-school and olympiad-style mathematics, yet their capabilities on advanced mathematics remain poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.11849","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_advplan-bench_7a0b9910","familyId":"bmf_9f699b02a07d","name":"AdvPlan-Bench","oneLine":"AdvPlan-Bench is an offline benchmark for adversarial evaluation of structured plan-generation agents. It uses typed action chains, adversarial response sets, selector diagnostics, and metrics like BLUE-vs-RED advantage and Nash-gap. Includes 150 synthetic scenarios across five planning templates.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00832","pdf":"https://arxiv.org/pdf/2608.00832","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00832"},"evidence":{"snippet":"We introduce AdvPlan-Bench, an offline benchmark for adversarial evaluation of structured plan-generation agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00832"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AdvPlan-Bench is an offline benchmark for adversarial evaluation of structured plan-generation agents. It uses typed action chains, adversarial response sets, selector diagnostics, and metrics like BLUE-vs-RED advantage and Nash-gap. Includes 150 synthetic scenarios across five planning templates.","whyItMatters":"Plan quality is often evaluated in isolation, but realistic tasks require considering adversarial responses. AdvPlan-Bench provides a reproducible method to study adversarial plan evaluation, response-budget sensitivity, and candidate frontiers, informing robust planning agent design.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ebfe1dc5a44777a398834bb733b3accb0e333e6639d5aca3978b5b5e7a682117"},"motivation":"Structured plan-generation agents are often evaluated as if a plan has quality in isolation, yet many realistic planning tasks require asking how a candidate behaves when another agent can search for responses.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00832","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ae-uav_5141a634","familyId":"bmf_df940847cd49","name":"AE-UAV","oneLine":"AE-UAV is an air-to-air event-based UAV tracking benchmark with 178 flight sequences and continuous-time cubic B-spline annotations, supporting evaluation at arbitrary temporal resolutions. It includes multimodal auxiliary data and predefined train/validation/test splits.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14726","pdf":"https://arxiv.org/pdf/2607.14726","project":null,"code":"https://github.com/MSP-xEN/AE-UAV","data":null,"hfPaper":"https://huggingface.co/papers/2607.14726"},"evidence":{"snippet":"To bridge these gaps, we introduce AE-UAV, an air-to-air event-based UAV tracking benchmark.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":12,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14726"},"ranking":{"90d":{"score":34,"rank":203,"coverage":0.7,"confidence":"Medium"}},"description":"AE-UAV is an air-to-air event-based UAV tracking benchmark with 178 flight sequences and continuous-time cubic B-spline annotations, supporting evaluation at arbitrary temporal resolutions. It includes multimodal auxiliary data and predefined train/validation/test splits.","whyItMatters":"This benchmark addresses the lack of dedicated event-based datasets for air-to-air UAV tracking, enabling consistent evaluation and comparison of tracking methods under diverse motion and illumination conditions. It provides a public resource for developing and validating real-time tracking solutions on resource-constrained platforms.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1b901fc1cfdb4aa8c50e87d37739d2b4c2123a033e310031c27f0f265213ba6c"},"motivation":"Air-to-air (A2A) unmanned aerial vehicle (UAV) tracking is fundamental to airborne remote sensing of low-altitude aerial targets.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14726","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AE-UAV Team","organizationType":"academic-lab","sourceUrl":"https://github.com/MSP-xEN/AE-UAV","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_07e09c5b14786ba9","familyId":"catalog_family_07e09c5b14786ba9","name":"ael_gate_benchmark_cases_template","oneLine":"Catalog-listed benchmark; original-source verification is pending.","description":"Catalog-listed benchmark; original-source verification is pending.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":[],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/community:5f95f778-c521-43fa-b80e-6a55465601e3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_07e09c5b14786ba9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/community:5f95f778-c521-43fa-b80e-6a55465601e3"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"community:5f95f778-c521-43fa-b80e-6a55465601e3","url":"https://llm-stats.com/benchmarks/community:5f95f778-c521-43fa-b80e-6a55465601e3","datasetSlug":"ael-gate-benchmark-cases-template","versionCount":1,"subsetCount":1,"rowCount":2,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":[],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_aerocopilotbench_7dd468f7","familyId":"bmf_cbfa21ffac06","name":"AeroCopilotBench","oneLine":"AeroCopilotBench is a two-tier benchmark for evaluating LLM agents as aviation copilots. Tier-1 uses 1,200 multiple-choice questions for knowledge assessment, while Tier-2 includes 73 procedural tasks in an interactive virtual cockpit environment, with safety-gated evaluation.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.16349","pdf":"https://arxiv.org/pdf/2608.16349","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.16349"},"evidence":{"snippet":"This paper presents the AeroCopilot Operational Environment (ACOE), a reproducible interactive virtual-cockpit test environment, and AeroCopilotBench, a two-tier aviation agent evaluation benchmark.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.16349"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AeroCopilotBench is a two-tier benchmark for evaluating LLM agents as aviation copilots. Tier-1 uses 1,200 multiple-choice questions for knowledge assessment, while Tier-2 includes 73 procedural tasks in an interactive virtual cockpit environment, with safety-gated evaluation.","whyItMatters":"Aviation evaluations often focus on static knowledge and cannot test procedural execution and safety compliance. AeroCopilotBench provides a reproducible interactive environment and safety-gated scoring to assess agents' task completion and trajectory safety.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1d2a2eb64ae7a62f2646b3d67615636af6a1bdce64fcbaa5a187bb30278a28ce"},"motivation":"Large language model (LLM) agents may assist flight crews with complex decisions and task execution, but existing aviation evaluations centered on static knowledge do not support systematic testing of procedural execution and safety compliance in interactive environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.16349","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_aeroground_3ddedf6a","familyId":"bmf_0dd007a3643a","name":"AeroGround","oneLine":"AeroGround evaluates vision-language models on aerial-ground collaborative reasoning using a simulated dataset of ~29,000 multimodal observation groups and 2,250 QA instances covering cross-view correspondence, spatial understanding, and reasoning.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.14721","pdf":"https://arxiv.org/pdf/2608.14721","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.14721"},"evidence":{"snippet":"To address this gap, we introduce AeroGround, a comprehensive benchmark for evaluating VLMs in aerial-ground collaborative reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14721"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AeroGround evaluates vision-language models on aerial-ground collaborative reasoning using a simulated dataset of ~29,000 multimodal observation groups and 2,250 QA instances covering cross-view correspondence, spatial understanding, and reasoning.","whyItMatters":"Existing UAV benchmarks focus on aerial-only views; AeroGround fills the gap for aerial-ground collaboration, offering a standardized evaluation for models in real-world applications like rescue and inspection, with clear human performance comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ebe3a22902cb80d2177531bb1b4faed54b18b91302869e46a93fd79adfcdedfe"},"motivation":"Vision-language models (VLMs) have been widely employed in understanding and reasoning tasks for unmanned aerial vehicles (UAVs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14721","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_5355c4d25a985352","familyId":"catalog_family_5355c4d25a985352","name":"AetherCode","oneLine":"AetherCode is a competitive-programming benchmark of olympiad-level algorithmic coding problems.","description":"AetherCode is a competitive-programming benchmark of olympiad-level algorithmic coding problems.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/aethercode","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5355c4d25a985352"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/aethercode"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"aethercode","url":"https://llm-stats.com/benchmarks/aethercode","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","coding"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_afdbench_6e3f5983","familyId":"bmf_041d5289a195","name":"AFDBench","oneLine":"Evaluates generative meteorological reasoning through 7,732 expert-written forecast discussions paired with AI weather inputs, using metrics for numerical accuracy, style, and grounding.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["AI Scientist","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-27","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.24954","pdf":"https://arxiv.org/pdf/2608.24954","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present AFDBench, an AI meteorologist that generates professional Area Forecast Discussions (AFDs) by reasoning through structured AI weather forecast data from Google's WeatherNext 2.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24954"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates generative meteorological reasoning through 7,732 expert-written forecast discussions paired with AI weather inputs, using metrics for numerical accuracy, style, and grounding.","whyItMatters":"Provides a high-stakes domain benchmark for factual accuracy and professional style in weather text generation, crucial for reliable AI-assisted communication.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"76dadb0743a4620b4b1334e34b76720063556b7d1afab5b8ea186ab53bcc989e"},"motivation":"Large language models (LLMs) hallucinate numerical values when generating high-stakes meteorological text, posing risks for weather communication.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"The paper presents a named benchmark with a dataset and three defined metrics, offering a stable evaluation setup for model comparison.","canonicalNameSource":"paper_title","canonicalNameEvidence":"AFDBench: A Reasoning-First AI Scientist for NationalWeather Service Forecast Discussions"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24954","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":70,"confidence":"Medium","horizon":"7d","reason":"Combines LLM evaluation with critical weather forecasting applications and reinforcement learning results, likely to draw attention from applied AI and meteorology communities."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_affordance20q_4a1ecb35","familyId":"bmf_b583dac6d78c","name":"AFFORDANCE20Q","oneLine":"Affordance20Q is a benchmark for evaluating affordance reasoning in LLMs using a 20-questions game. It comprises 1,009 games over 454 objects and 59 affordances, where models identify a hidden object's affordance by asking yes/no questions about physical properties.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.14240","pdf":"https://arxiv.org/pdf/2606.14240","project":null,"code":"https://github.com/1171-jpg/Affordance20Q.git","data":null,"hfPaper":"https://huggingface.co/papers/2606.14240"},"evidence":{"snippet":"To address this gap, we introduce Affordance20Q, a novel affordance reasoning benchmark formulated as a 20-Questions game without exposing the object's identity.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":"2026-06-15T00:00:00.000Z","githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.14240"},"ranking":{"90d":{"score":20,"rank":388,"coverage":0.7,"confidence":"Medium"}},"description":"Affordance20Q is a benchmark for evaluating affordance reasoning in LLMs using a 20-questions game. It comprises 1,009 games over 454 objects and 59 affordances, where models identify a hidden object's affordance by asking yes/no questions about physical properties.","whyItMatters":"Affordance reasoning is fundamental to physical understanding. Affordance20Q tests whether models can reason over physical properties without relying on memorized object-affordance mappings, which is crucial for embodied AI.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dd61f2b3753053a683d5b5fa8b53ec7f3847c5984b3afc0b06672e4172a4b092"},"motivation":"Affordance reasoning, the inference of an object's action possibilities from its physical properties (e.g., shape and material), is fundamental to human physical understanding and increasingly critical for Large Language Models (LLMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.14240","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_callability-is-not-operability-controlled-_13caf3a3","familyId":"bmf_3d31407ac492","name":"AFT-Bench","oneLine":"Holds task, backend, initial state, injected failure, agent, and language model fixed while varying the tool interface to measure callability versus operability.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-26","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.23628","pdf":"https://arxiv.org/pdf/2608.23628","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce AFT-Bench, a controlled interface-intervention framework that holds the task, backend, initial state, injected failure, agent, and language model fixed while varying the interface exposed to the agent.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23628"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Holds task, backend, initial state, injected failure, agent, and language model fixed while varying the tool interface to measure callability versus operability.","whyItMatters":"It isolates interface-level causes of agent failures and enables controlled experiments on how tool APIs affect safe continuation.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"fba5937c65e368708790824ca0cbdf5cfc932caded88e9d6a9aa4f8360739480"},"motivation":"A tool call can be perfectly valid yet still leave an autonomous agent unable to determine what to do next.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The framework provides a repeatable protocol and is designed as a resource other teams can reuse for interface-intervention studies.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce AFT-Bench, a controlled interface-intervention framework"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.23628","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":64,"confidence":"Medium","horizon":"7d","reason":"Controlled intervention benchmarks for LLM agent tool usage are timely, though the abstract alone may limit immediate visibility."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_agc-bench_28cf2135","familyId":"bmf_8f34ce070e53","name":"AGC-Bench","oneLine":"AGC-Bench evaluates artificial general creativity across 78 datasets covering brainstorming, problem solving, STEM, narrative, figurative language, and humor. It uses an agentic harness and a public leaderboard with an open-weight judge model.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.01152","pdf":"https://arxiv.org/pdf/2607.01152","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.01152"},"evidence":{"snippet":"We introduce AGC-Bench, an artificial general creativity benchmark built from a systematic review of the AI creativity literature (3,101 papers screened, 497 benchmarks identified), paired with an agentic harness that converts idiosyncratic codebases into HELM-standardized benchmarks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01152"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AGC-Bench evaluates artificial general creativity across 78 datasets covering brainstorming, problem solving, STEM, narrative, figurative language, and humor. It uses an agentic harness and a public leaderboard with an open-weight judge model.","whyItMatters":"Creativity is a key aspect of intelligence often overlooked in AI evaluation. AGC-Bench provides a comprehensive, standardized measurement of AI creativity, revealing distinct strengths and weaknesses across models and a single creativity factor analogous to general intelligence.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dd67d7d201c57369233f4f7573d29aacdfb2b30bc95a1bd759c228e22bc41d54"},"motivation":"Creativity research has debated whether creativity is domain-specific (e.g., visual, writing, science), and if it is psychometrically separable from general intelligence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01152","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agent-app-benchmark_0bc5ce52","familyId":"bmf_407a33ad9273","name":"Agent App Benchmark","oneLine":"Evaluates GUI application performance for coding agents using deterministic historical session workloads, measuring app start, session switching, memory, and CPU metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Agents"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-24","recognitionConfidence":0.4,"links":{"report":"https://github.com/kyashrathore/agent-app-benchmark","pdf":null,"project":null,"code":"https://github.com/kyashrathore/agent-app-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"agent-app-benchmark Reproducible performance benchmarks for multi-harness coding-agent GUI applications # Agent App Benchmark Agent App Benchmark is a public, reproducible performance benchmark for multi-harness coding-agent GUI applications.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:kyashrathore/agent-app-benchmark"},"ranking":{"30d":{"score":23,"rank":130,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":334,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates GUI application performance for coding agents using deterministic historical session workloads, measuring app start, session switching, memory, and CPU metrics.","whyItMatters":"Provides a reproducible performance comparison for coding-agent GUI applications, focusing on responsiveness and resource usage under realistic session load.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"81b409d6b38d5fa7c166abf66f90652fda0f5064884ceb6a2df2431273d0e3d9"},"motivation":"agent-app-benchmark Reproducible performance benchmarks for multi-harness coding-agent GUI applications # Agent App Benchmark Agent App Benchmark is a public, reproducible performance benchmark for multi-harness coding-agent GUI applications.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/kyashrathore/agent-app-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-24T07:40:52.396154Z"},"attentionForecast":{"score":28,"confidence":"Low","horizon":"7d","reason":"The benchmark addresses a niche but emerging area of coding-agent GUI performance, though it lacks independent publications or widespread adoption evidence."},"evaluationMode":"score_submission","publishers":[{"name":"kyashrathore","organizationType":"community","sourceUrl":"https://github.com/kyashrathore/agent-app-benchmark","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_agent-planning-benchmark_ba86aa47","familyId":"bmf_13413c2da00d","name":"Agent Planning Benchmark","oneLine":"Agent Planning Benchmark (APB) is a diagnostic benchmark with 4,209 multimodal cases across 22 domains, evaluating planning capabilities in five settings including tool noise and unsolvable tasks.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning","Robustness"],"topics":["Agents","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04874","pdf":"https://arxiv.org/pdf/2606.04874","project":null,"code":"https://github.com/Mikivishy/AgentPlanningBenchmark","data":null,"hfPaper":"https://huggingface.co/papers/2606.04874"},"evidence":{"snippet":"We introduce Agent Planning Benchmark (APB), a planning-specific diagnostic benchmark with 4,209 multimodal cases across 22 domains and five settings, covering holistic planning, feedback-conditioned step-wise planning, and robustness under extraneous tools, broken tools, and unsolvable tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04874"},"ranking":{"90d":{"score":30,"rank":247,"coverage":0.7,"confidence":"Medium"}},"description":"Agent Planning Benchmark (APB) is a diagnostic benchmark with 4,209 multimodal cases across 22 domains, evaluating planning capabilities in five settings including tool noise and unsolvable tasks.","whyItMatters":"Provides a planning-specific evaluation that isolates failures from execution, enabling targeted improvement of agent planning and refusal behaviors.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ce2467c3e2d3a7e58a80d61c03d04197e43368639a0b6597576cb2b69aea1082"},"motivation":"Planning is central to LLM agents: before acting, an agent must decompose goals, select tools, reason over constraints, and decide when a task is infeasible.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04874","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"APB Team","organizationType":"academic-lab","sourceUrl":"https://github.com/Mikivishy/AgentPlanningBenchmark","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_ade3a3de7a3f13a9","familyId":"catalog_family_ade3a3de7a3f13a9","name":"Agent Poker Bench","oneLine":"Which model can make the most money playing poker?","description":"Which model can make the most money playing poker?","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/poker_agent","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ade3a3de7a3f13a9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/pokeragent"}],"catalogSources":[{"catalog":"benchlm","sourceId":"pokerAgent","url":"https://benchlm.ai/benchmarks/pokeragent","paperUrl":"https://www.vals.ai/benchmarks/poker_agent","year":"2026","fullName":"Vals Agent Poker Bench","format":"Accuracy score","tasks":"Poker-playing agent trials","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agent-retrieval-bench_1105bc27","familyId":"bmf_0f8de99c2972","name":"Agent Retrieval Bench","oneLine":"File-level retrieval benchmark for coding agents, covering four positive tasks (code2test, comment2context, trace2code, edit2ripple) and a selective-retrieval subset with natural no-gold and counterfactual controls across 25 repositories, 427 samples, with frozen base-commit corpora.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Information retrieval"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.24882","pdf":"https://arxiv.org/pdf/2607.24882","project":null,"code":"https://github.com/eyuansu62/agent-retrieval-bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.24882"},"evidence":{"snippet":"We introduce Agent Retrieval Bench, a file-level benchmark for this upstream retrieval problem.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":6,"hfDailySubmittedAt":null,"githubStars":7,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24882"},"ranking":{"90d":{"score":36,"rank":190,"coverage":0.7,"confidence":"Medium"}},"description":"File-level retrieval benchmark for coding agents, covering four positive tasks (code2test, comment2context, trace2code, edit2ripple) and a selective-retrieval subset with natural no-gold and counterfactual controls across 25 repositories, 427 samples, with frozen base-commit corpora.","whyItMatters":"Provides a dedicated evaluation for the context-acquisition stage of coding agents, distinguishing retrieval quality from patch generation and offering a reusable protocol for comparing retrieval and selective abstention methods.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4729577774258c87bd81952797c80bdace3b32993b4c2e0c7e1075a47f3acc68"},"motivation":"Modern coding agents are usually evaluated by whether they eventually produce a correct patch, but patch generation depends on an earlier context-acquisition stage: finding the repository files needed for the task.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24882","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Agent Retrieval Bench team","organizationType":"academic-lab","sourceUrl":"https://github.com/eyuansu62/agent-retrieval-bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval","Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_06a52f8a0bca38b1","familyId":"catalog_family_06a52f8a0bca38b1","name":"Agent Startup Bench","oneLine":"Agent Startup Bench measures AI agents on high-economic-value, startup-style tasks that require autonomous planning and execution to deliver practical, verifiable results.","description":"Agent Startup Bench measures AI agents on high-economic-value, startup-style tasks that require autonomous planning and execution to deliver practical, verifiable results.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/agent-startup-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_06a52f8a0bca38b1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/agent-startup-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"agent-startup-bench","url":"https://llm-stats.com/benchmarks/agent-startup-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_agent-model-bench_4106c088","familyId":"bmf_edbdc48fb887","name":"agent-model-bench","oneLine":"Evaluates language models on tool calling and strict JSON adherence using 32 cases, scoring exact tool and argument matches and JSON parsing with strict and content-based criteria.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-21","firstSeenAt":"2026-08-22","recognitionConfidence":0.4,"links":{"report":"https://github.com/ajbermudezh22/agent-model-bench","pdf":null,"project":"https://ajbermudezh22.github.io/agent-model-bench/","code":"https://github.com/ajbermudezh22/agent-model-bench","data":null,"hfPaper":null},"evidence":{"snippet":"agent-model-bench Which model should power your agent?","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:ajbermudezh22/agent-model-bench"},"ranking":{"30d":{"score":23,"rank":137,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":341,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates language models on tool calling and strict JSON adherence using 32 cases, scoring exact tool and argument matches and JSON parsing with strict and content-based criteria.","whyItMatters":"Offers a small, honest screen for agent-critical capabilities, exposing formatting discipline differences that practical JSON parsing would penalize.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"9e921b5a1165c6183814bf60f81844bba971f83781ac686c60da848da065c65f"},"motivation":"agent-model-bench Which model should power your agent?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/ajbermudezh22/agent-model-bench","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-22T19:19:35.289219Z"},"attentionForecast":{"score":35,"confidence":"Low","horizon":"7d","reason":"The small case count and narrow scope limit visibility, though the focus on tool calling and JSON faithfulness addresses a recognized need in agent frameworks."},"evaluationMode":"public_reusable","publishers":[{"name":"ajbermudezh22","organizationType":"community","sourceUrl":"https://github.com/ajbermudezh22/agent-model-bench","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agentabstain_d94f5610","familyId":"bmf_4b74dce329f5","name":"AgentAbstain","oneLine":"AgentAbstain is a paired-task benchmark for evaluating LLM agents' ability to abstain from acting in scenarios such as ambiguity, conflicting constraints, or tool failures. It includes 263 paired tasks across 42 sandbox environments, with a proposed pipeline for generating fresh task instances.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Agents","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.10059","pdf":"https://arxiv.org/pdf/2607.10059","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.10059"},"evidence":{"snippet":"At its core, AgentAbstain is a paired-task benchmark built on an agent-native taxonomy of 8 abstention scenarios across pre-execution reasoning and runtime discovery.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.10059"},"ranking":{"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"AgentAbstain is a paired-task benchmark for evaluating LLM agents' ability to abstain from acting in scenarios such as ambiguity, conflicting constraints, or tool failures. It includes 263 paired tasks across 42 sandbox environments, with a proposed pipeline for generating fresh task instances.","whyItMatters":"Agent abstention is critical for safe deployment, yet existing evaluations focus on task success. This benchmark targets the gap in measuring calibrated abstention, highlighting that abstention capability is independent of general task-solving ability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"72e3cdd0dcd558b7f52a77710bf5c5d8655a5f4569db27a6fc48214e0bed26dd"},"motivation":"Agent systems based on large language models (LLMs) are increasingly deployed for autonomous tasks, yet existing evaluations mostly focus on task success rather than whether agents know when to abstain.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.10059","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agentchaosbench_3680a518","familyId":"bmf_6edc4728dc4b","name":"AGENTCHAOSBENCH","oneLine":"AGENTCHAOSBENCH is a dataset of sanitized execution traces from five agentic applications with injected runtime faults, used to evaluate fault detection and localization from telemetry.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.14680","pdf":"https://arxiv.org/pdf/2608.14680","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.14680"},"evidence":{"snippet":"We present AGENTCHAOSBENCH, a benchmark for detecting and localizing runtime faults in agentic systems from their execution telemetry.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14680"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AGENTCHAOSBENCH is a dataset of sanitized execution traces from five agentic applications with injected runtime faults, used to evaluate fault detection and localization from telemetry.","whyItMatters":"Addresses runtime fault diagnosis in LLM agentic systems, offering a reproducible task for comparing diagnostic methods across tool and agent boundaries.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d8265f900a0f5fb1b712b381254c06eae813c895964a23a9b7412f07c0e9c5c4"},"motivation":"Reliability in LLM-based agentic systems is a property of the whole execution (its tool calls, model calls, guardrails, and inter-agent messages), not of the final answer alone, yet evaluating only task outcomes reveals little about how or why a run fails.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14680","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agentfairbench_1a8330ea","familyId":"bmf_806b46eeb591","name":"AgentFairBench","oneLine":"AgentFairBench evaluates demographic disparity in the actions of LLM agents across hiring, lending, and medical triage. It uses synthetic, demographic-neutral profiles in counterfactual matched sets varying name-coded race/gender. Metrics include counterfactual flip rate, mean absolute score difference, action-rate disparity, and tool-invocation disparity, with bootstrap confidence intervals and FDR control.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.16723","pdf":"https://arxiv.org/pdf/2606.16723","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.16723"},"evidence":{"snippet":"We introduce AgentFairBench, a cheap, reproducible, multi-domain benchmark for demographic disparity in the actions of LLM agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16723"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"AgentFairBench evaluates demographic disparity in the actions of LLM agents across hiring, lending, and medical triage. It uses synthetic, demographic-neutral profiles in counterfactual matched sets varying name-coded race/gender. Metrics include counterfactual flip rate, mean absolute score difference, action-rate disparity, and tool-invocation disparity, with bootstrap confidence intervals and FDR control.","whyItMatters":"Existing fairness evaluations grade answers, not actions. AgentFairBench addresses the gap by measuring disparity in consequential agent decisions. Its low cost and reproducible harness provide a practical path for screening models for action-level bias before deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"867d877722466981e8b6a05208f9e754896d1d37100e8809e931de311d2e2aa0"},"motivation":"Large language model (LLM) agents increasingly take actions (screening applicants, recommending credit, triaging patients), yet fairness for LLMs is still measured by grading answers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"publication_reported","venue":"Under Review (2026)","evidence":"Under Review (2026)","evidenceUrl":"https://arxiv.org/abs/2606.16723","source":"arxiv-journal-reference","evidenceLevel":"strong-author-metadata","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publications":[{"venueName":"Under Review (2026)","publicationStatus":"published","evidence":[{"sourceType":"arxiv-journal-reference","sourceUrl":"https://arxiv.org/abs/2606.16723","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Under Review (2026)","level":"strong-author-metadata"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agenthijack_15cbe4d0","familyId":"bmf_48bd013b2a40","name":"AgentHijack","oneLine":"Evaluates the robustness of computer use agents under common environment corruptions such as pop-ups, resolution changes, and competing applications. The benchmark introduces 9 configurable corruptions and evaluates agent performance on desktop tasks using multimodal LLM-based agents, measuring task completion rates.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Robustness"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.25707","pdf":"https://arxiv.org/pdf/2605.25707","project":"https://AgentHijack.github.io","code":"https://github.com/tmlr-group/AgentHijack","data":null,"hfPaper":"https://huggingface.co/papers/2605.25707"},"evidence":{"snippet":"We introduce AgentHijack, a benchmark designed to evaluate the robustness of computer-use agents under common corruptions, where the uncertainties in dynamic environment disrupt the execution flow without direct adversarial intent.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":6,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25707"},"ranking":{},"description":"Evaluates the robustness of computer use agents under common environment corruptions such as pop-ups, resolution changes, and competing applications. The benchmark introduces 9 configurable corruptions and evaluates agent performance on desktop tasks using multimodal LLM-based agents, measuring task completion rates.","whyItMatters":"Real-world execution environments are imperfect, and minor corruptions can cause significant performance degradation. This benchmark quantifies agent fragility and supports the development of more robust computer use agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8eec12423b8a25aee11bc508fa90f3d94f57674ede97c647b5518cf304495aa3"},"motivation":"Autonomous computer use agents that powered by multimodal large language models (MLLMs) are emerging as capable assistants for completing complex digital workflows.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026","evidence":"accepted by ICML 2026","evidenceUrl":"https://arxiv.org/abs/2605.25707","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026","reviewStatus":"accepted","decisionRaw":"accepted by ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2605.25707","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"accepted by ICML 2026","level":"author-claim"}]}],"publishers":[{"name":"AgentHijack Team","organizationType":"academic-lab","sourceUrl":"https://github.com/tmlr-group/AgentHijack","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_agenthpobench_61915136","familyId":"bmf_c788cbfe6bdb","name":"AgentHPOBench","oneLine":"AgentHPOBench evaluates LLM agents as sequential hyperparameter optimizers across 30 executable ML tasks. Agents observe accumulated configurations, metrics, and logs, then propose the next configuration. Scoring compares agents and conventional HPO baselines under a unified protocol.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.29626","pdf":"https://arxiv.org/pdf/2607.29626","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.29626"},"evidence":{"snippet":"To address this gap, we introduce AgentHPOBench, a sequential benchmark comprising 30 executable machine learning tasks across seven research categories.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.29626"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AgentHPOBench evaluates LLM agents as sequential hyperparameter optimizers across 30 executable ML tasks. Agents observe accumulated configurations, metrics, and logs, then propose the next configuration. Scoring compares agents and conventional HPO baselines under a unified protocol.","whyItMatters":"Existing benchmarks do not directly assess whether agents can interpret experimental evidence and use it to guide subsequent hyperparameter decisions. AgentHPOBench provides a repeatable protocol for measuring iterative decision-making, filling a gap in evaluating autonomous scientific agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6e96b78ef967b4aaa652baba8aedfc43c3141d44ed9ae91a41f1399a880dedae"},"motivation":"As LLMs evolve from code completion systems into autonomous scientific agents, evaluating their ability to conduct experiments has become increasingly important.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.29626","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AgentHPOBench team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.29626","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agent-benchmark-2d-maze_4d511c00","familyId":"bmf_d82de6e84a4d","name":"Agentic Evaluation in 2D Mazes","oneLine":"Evaluates interactive multimodal agents in structured 2D maze environments requiring long-horizon planning, mechanism interaction, and error recovery, scored by mechanism-aware progress.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Agents","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/bluebunnyanon/agent-benchmark-2d-maze","pdf":null,"project":null,"code":"https://github.com/bluebunnyanon/agent-benchmark-2d-maze","data":null,"hfPaper":null},"evidence":{"snippet":"agent-benchmark-2d-maze # Agentic Evaluation in 2D Mazes This repository contains the code, benchmark instances, evaluation harness, and reproducibility artifacts accompanying an anonymous submission on evaluating interactive multimodal agents in structured 2D maze environments.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:bluebunnyanon/agent-benchmark-2d-maze"},"ranking":{"30d":{"score":23,"rank":129,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":333,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates interactive multimodal agents in structured 2D maze environments requiring long-horizon planning, mechanism interaction, and error recovery, scored by mechanism-aware progress.","whyItMatters":"Provides a controlled testbed for agent capabilities beyond single-step tasks, with a human-playable interface and modular harness for model comparison.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"0ded56072cac6d4ce03a7c2e9308e8a11650b04b56b4c9c513b6347e59ac0e37"},"motivation":"agent-benchmark-2d-maze # Agentic Evaluation in 2D Mazes This repository contains the code, benchmark instances, evaluation harness, and reproducibility artifacts accompanying an anonymous submission on evaluating interactive multimodal agents in structured 2D maze environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/bluebunnyanon/agent-benchmark-2d-maze","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":45,"confidence":"Low","horizon":"7d","reason":"The benchmark addresses a growing area in agent evaluation, but its anonymous nature and niche 2D maze focus may limit early visibility."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Agents","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_agenticdatabench_3427dc68","familyId":"bmf_d78a1404db17","name":"AgenticDataBench","oneLine":"Evaluates LLM-based data agents on realistic data science workflows across 15 domains, with fine-grained ground-truth labels and skill-level scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.DB"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.01647","pdf":"https://arxiv.org/pdf/2607.01647","project":null,"code":"https://github.com/AgenticDataBench/AgenticDataBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.01647"},"evidence":{"snippet":"To address this gap, we propose AgenticDataBench, a comprehensive benchmark featuring realistic tasks spanning diverse domains with fine-grained ground-truth labels.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":38,"hfDailySubmittedAt":"2026-07-03T00:00:00.000Z","githubStars":39,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01647"},"ranking":{"90d":{"score":52,"rank":50,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates LLM-based data agents on realistic data science workflows across 15 domains, with fine-grained ground-truth labels and skill-level scoring.","whyItMatters":"Addresses the lack of comprehensive benchmarks for automating data science workflows, providing fine-grained skill-level insights to guide agent development and selection.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d6074631e8e0d1c9c5df13ee62df0befc8c9731a23deea6cb4565001a4107055"},"motivation":"Data science aims to derive actionable insights from heterogeneous raw data, unlocking the value of the massive amounts of data generated in modern society.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01647","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agenticinterpbench_4e5baf78","familyId":"bmf_6546f2911c91","name":"AgenticInterpBench","oneLine":"Evaluates language model agents on explaining components of transformer circuits, with 84 semi-synthetic circuits and 163 component-level annotations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24026","pdf":"https://arxiv.org/pdf/2606.24026","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.24026"},"evidence":{"snippet":"We introduce AgenticInterpBench, a benchmark for circuit explanation built from 84 semi-synthetic transformer circuits with 163 component-level annotations.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24026"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates language model agents on explaining components of transformer circuits, with 84 semi-synthetic circuits and 163 component-level annotations.","whyItMatters":"Addresses the lack of standardized evaluation for circuit explanation in mechanistic interpretability, but lacks a standalone public comparison path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bef0db407df995cb074583af471d5bf49e108f89564c797979c2fa8eee814c2f"},"motivation":"Mechanistic interpretability has made substantial progress in automatically localizing circuits, but explaining what localized components do remains labor-intensive and difficult to standardize.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24026","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agentjudgebench_8ae8da3e","familyId":"bmf_cfd6a96e803e","name":"AgentJudgeBench","oneLine":"LLM judges are widely used to evaluate agentic tool-calling systems, yet their reliability on structured, dependency-driven workflows remains largely unexamined.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.26623","pdf":"https://arxiv.org/pdf/2608.26623","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.26623"},"evidence":{"snippet":"We present AgentJudgeBench, the first benchmark to systematically study LLM-as-a-judge reliability for agentic tool-calling over workflow DAGs, as distinct from the broader LLM-as-a-judge task of open-ended text or preference evaluation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26623"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"LLM judges are widely used to evaluate agentic tool-calling systems, yet their reliability on structured, dependency-driven workflows remains largely unexamined.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"acceptance_claimed","venue":"as a main conference paper at EMNLP 2026","evidence":"31 Pages, 9 Figures, 31 Tables, Accepted as a main conference paper at EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2608.26623","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"as a main conference paper at EMNLP 2026","reviewStatus":"accepted","decisionRaw":"31 Pages, 9 Figures, 31 Tables, Accepted as a main conference paper at EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.26623","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"31 Pages, 9 Figures, 31 Tables, Accepted as a main conference paper at EMNLP 2026","level":"author-claim"}]}],"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agentlens_07b9cd48","familyId":"bmf_95050afca81d","name":"AgentLens","oneLine":"AgentLens evaluates interactive coding agents across entire trajectories, combining formal verification with LLM-written trajectory reviews and side-by-side comparisons to score dimensions such as instruction compliance, tool use, and interaction style.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.06624","pdf":"https://arxiv.org/pdf/2607.06624","project":null,"code":"https://github.com/agent-lens/agent-lens-bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.06624"},"evidence":{"snippet":"We present AgentLens, a production-assessed benchmark for interactive code agents.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":9,"hfDailySubmittedAt":"2026-07-09T00:00:00.000Z","githubStars":8,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06624"},"ranking":{"90d":{"score":38,"rank":167,"coverage":0.7,"confidence":"Medium"}},"description":"AgentLens evaluates interactive coding agents across entire trajectories, combining formal verification with LLM-written trajectory reviews and side-by-side comparisons to score dimensions such as instruction compliance, tool use, and interaction style.","whyItMatters":"Most code-agent benchmarks reduce an episode to a single pass/fail bit, which is too coarse for production use. AgentLens provides a richer evaluation that captures real user experience and yields readable justifications for scores, aiding regression detection and model diagnosis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8c31964b90f1f6e7fe19a4da453e193bd6736a323c42fc0e7a19f051516cc381"},"motivation":"We present AgentLens, a production-assessed benchmark for interactive code agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06624","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_agentmembench_9938dbed","familyId":"bmf_9d9e93f978b2","name":"AgentMemBench","oneLine":"AgentMemBench evaluates five long-term memory management strategies for conversational AI agents across three public datasets (LoCoMo, MultiDoc2Dial, MSC), covering multi-session dialogue, document grounding, and persona-grounded chat. Scoring uses retrieval metrics (Recall@k, MRR, nDCG@k), Answer F1, LLM-judge faithfulness, memory footprint, and latency over 491 annotated question turns.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00009","pdf":"https://arxiv.org/pdf/2608.00009","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00009"},"evidence":{"snippet":"We present AgentMemBench, a unified, reproducible benchmark evaluating five memory management strategies under identical conditions: in-context windowing (ICW), external key-value store (EKV), graph-based episodic memory (GEM), compression-based summarisation (CBS), and web-augmented memory (WAM).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00009"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AgentMemBench evaluates five long-term memory management strategies for conversational AI agents across three public datasets (LoCoMo, MultiDoc2Dial, MSC), covering multi-session dialogue, document grounding, and persona-grounded chat. Scoring uses retrieval metrics (Recall@k, MRR, nDCG@k), Answer F1, LLM-judge faithfulness, memory footprint, and latency over 491 annotated question turns.","whyItMatters":"Long-term memory is a key bottleneck for conversational agents. AgentMemBench provides a controlled, reproducible comparison of memory strategies under identical conditions, enabling practitioners to make informed trade-offs between recall quality, accuracy, and resource cost.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"106686e4a9d95585a33c4ef011227553ef30c929aa78e88f7784e6e26e67e0b9"},"motivation":"Long-term memory remains a critical bottleneck for conversational AI agents, whose finite context windows cannot support coherent recall across thousands of turns.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00009","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AgentMemBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.00009","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agentredbench_2025428a","familyId":"bmf_1ff4fb25b17b","name":"AgentRedBench","oneLine":"AgentRedBench evaluates LLM agents against indirect prompt injection and underspecified-authorization attacks across 24 enterprise SaaS integrations. It defines 215 attack scenarios with immutable versioning, and tracks attack success rate (ASR) for models and defenses. Open-source codebase and schemas enable replay.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CR"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02240","pdf":"https://arxiv.org/pdf/2606.02240","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02240"},"evidence":{"snippet":"We introduce AGENTREDBENCH, a dynamic LLM-driven redteaming benchmark of 215 subtle underspecified-authorization scenarios across 24 enterprise integrations and five attack types.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02240"},"ranking":{},"description":"AgentRedBench evaluates LLM agents against indirect prompt injection and underspecified-authorization attacks across 24 enterprise SaaS integrations. It defines 215 attack scenarios with immutable versioning, and tracks attack success rate (ASR) for models and defenses. Open-source codebase and schemas enable replay.","whyItMatters":"Prior agent-security benchmarks cover limited integrations with static payloads, understating real-world exploitability. AgentRedBench provides a dynamic, maintainer-mediated evaluation to compare model and guard-rail effectiveness against evolving injection threats, supporting procurement and hardening decisions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"589fc791f1a241604065026b182fcb2f17bb478b63f04e4d447c9ec10588c3de"},"motivation":"Indirect prompt injection in tool-use agents is a concrete production threat: LLM agents read from integrations (third-party services such as Gmail, Salesforce, or Jira accessed through tool calls) whose response content the user neither writes nor controls.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02240","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agentrelbench_829efc3a","familyId":"bmf_4093d9a886cf","name":"AgentRelBench","oneLine":"AgentRelBench measures repeated agent runs in a fixed task suite, computing severity-weighted damage from database state diffs with no LLM judging.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-15","firstSeenAt":"2026-08-19","recognitionConfidence":0.75,"links":{"report":"https://arxiv.org/abs/2608.15286","pdf":"https://arxiv.org/pdf/2608.15286","project":null,"code":"https://github.com/shivenkk/agentrelbench","data":null,"hfPaper":null},"evidence":{"snippet":"We introduce AgentRelBench, an environment-agnostic reliability instrument that computes ground-truth, severity-priced damage from database state diffs across repeated runs, with no LLM in the measurement path, demonstrated on EnterpriseOps-Gym.","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15286"},"ranking":{"30d":{"score":28,"rank":95,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":265,"coverage":0.55,"confidence":"Low"}},"description":"AgentRelBench measures repeated agent runs in a fixed task suite, computing severity-weighted damage from database state diffs with no LLM judging.","whyItMatters":"It provides ground-truth reliability signals that evade single-run safety audits, helping evaluators detect stochastic agent failures.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"527e400505b67cd36108dc4fcaba92b4bc341ae050a81f3dd1c6cb8850f881e3"},"motivation":"We introduce AgentRelBench, an environment-agnostic reliability instrument that computes ground-truth, severity-priced damage from database state diffs across repeated runs, with no LLM in the measurement path, demonstrated on EnterpriseOps-Gym.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"A pre-registered, reproducible protocol with released code, task suite, per-run verdicts, and a clearly defined scoring method qualifies as a formal benchmark with a public reuse path.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce AgentRelBench, an environment-agnostic reliability instrument that computes ground-truth, severity-priced damage from database state diffs across repeated runs"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15286","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":62,"confidence":"Medium","horizon":"7d","reason":"Multi-model agent reliability audits with reproducible artifacts and a clear safety focus are likely to draw moderate immediate interest."},"evaluationMode":"public_reusable","publishers":[{"name":"AgentRelBench team","organizationType":"benchmark-organization","sourceUrl":"https://github.com/shivenkk/agentrelbench","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_d0b03ebbcb611ba0","familyId":"catalog_family_d0b03ebbcb611ba0","name":"Agents' Last Exam","oneLine":"Agents' Last Exam is a challenging benchmark for AI agents on hard, long-horizon tasks that test sustained reasoning, planning, and tool use, reported with and without tool access.","description":"Agents' Last Exam is a challenging benchmark for AI agents on hard, long-horizon tasks that test sustained reasoning, planning, and tool use, reported with and without tool access.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agentic","Reasoning","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://api-docs.deepseek.com/zh-cn/updates/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d0b03ebbcb611ba0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/agentslastexam"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/agents-last-exam"}],"catalogSources":[{"catalog":"benchlm","sourceId":"agentsLastExam","url":"https://benchlm.ai/benchmarks/agentslastexam","paperUrl":"https://api-docs.deepseek.com/zh-cn/updates/","year":"2026","fullName":"Agents' Last Exam","format":"Provider-reported task score","tasks":"Agent tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"agents-last-exam","url":"https://llm-stats.com/benchmarks/agents-last-exam","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","agents","tool calling"],"catalogModelCount":13,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_agents4d_33599565","familyId":"bmf_0ea9b8cc532e","name":"AgentS4D","oneLine":"AgentS4D evaluates runtime safety of LLM-based workspace agents across a four-dimensional framework, with 328 risk-injected cases and seven lifecycle checkpoints, measuring unsafe behavior and evidence across six risk-entry sources and nine harms.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27294","pdf":"https://arxiv.org/pdf/2607.27294","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.27294"},"evidence":{"snippet":"We introduce AgentS4D, a sandboxed benchmark for lifecycle-wide runtime safety evaluation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27294"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AgentS4D evaluates runtime safety of LLM-based workspace agents across a four-dimensional framework, with 328 risk-injected cases and seven lifecycle checkpoints, measuring unsafe behavior and evidence across six risk-entry sources and nine harms.","whyItMatters":"Existing safety benchmarks focus on endpoints, missing risks that emerge during execution. This benchmark provides a structured way to assess agent safety across the lifecycle, showing that task completion does not imply safety and that testing one risk form can miss vulnerabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d2931d4ef15987426df2a5fd3166b316d072f59bbad4c5d4aa4a9a79e91534bc"},"motivation":"Large language model (LLM)-based workspace agents execute stateful, multi-step workflows across heterogeneous resources, external tools, and persistent state.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27294","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_agentsysbench_213ca6e5","familyId":"bmf_ff8fb088901b","name":"AgentSysBench","oneLine":"AgentSysBench is a benchmark suite and measurement toolkit for characterizing agentic workloads on LLM serving systems. It includes ten representative agentic applications and unified instrumentation, identifying six properties that distinguish agentic workloads from conventional LLM inference.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.OS"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15127","pdf":"https://arxiv.org/pdf/2608.15127","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.15127"},"evidence":{"snippet":"We present AgentSysBench, a benchmark suite and measurement toolkit with ten representative agentic applications and unified systems-level instrumentation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15127"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AgentSysBench is a benchmark suite and measurement toolkit for characterizing agentic workloads on LLM serving systems. It includes ten representative agentic applications and unified instrumentation, identifying six properties that distinguish agentic workloads from conventional LLM inference.","whyItMatters":"Serving systems are designed for conventional LLM inference and may not handle agentic workloads efficiently. AgentSysBench provides a measurement-based characterization that could guide system design for agentic applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"669ded87be577731c12b75ef463ae18277f70618c04fa3be18502ca2e58a8632"},"motivation":"Agentic applications are shifting AI serving from isolated model inference to long-running workloads in which LLMs coordinate tools, environments, and persistent state.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15127","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agentworldbench_775e3273","familyId":"bmf_2508b6fd7ca7","name":"AgentWorldBench","oneLine":"AgentWorldBench evaluates language world models on simulation fidelity across 7 domains (MCP, Search, Terminal, SWE, Android, Web, OS) using real-world trajectories and rubric-based scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.24597","pdf":"https://arxiv.org/pdf/2606.24597","project":null,"code":"https://github.com/QwenLM/Qwen-AgentWorld","data":"https://huggingface.co/datasets/Qwen/AgentWorldBench","hfPaper":"https://huggingface.co/papers/2606.24597"},"evidence":{"snippet":"To evaluate language world models, we present AgentWorldBench, a comprehensive benchmark constructed from real-world interactions of 5 frontier models on 9 established benchmarks.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":159,"hfDailySubmittedAt":"2026-06-24T00:00:00.000Z","githubStars":982,"githubScope":"benchmark_repo","hfDatasetDownloads":1448,"hfDatasetLikes":103},"source":{"type":"arxiv","id":"2606.24597"},"ranking":{"90d":{"score":83,"rank":3,"coverage":1.0,"confidence":"High","datasetDownloadRank":15,"datasetRankPopulation":66}},"description":"AgentWorldBench evaluates language world models on simulation fidelity across 7 domains (MCP, Search, Terminal, SWE, Android, Web, OS) using real-world trajectories and rubric-based scoring.","whyItMatters":"This benchmark addresses the lack of systematic evaluation for language world models in agentic environments, providing a standardized protocol that enables direct comparison and guides development of general agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d2f0149d859e2cc3718f357e72bdc1a4b58e0d7c89f4f4e6cdffa38eb0cc7176"},"motivation":"A world model predicts environment dynamics based on current observations and actions, serving as a core cognitive mechanism for reasoning and planning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"source-reviewed","reviewedAt":"2026-08-30","sources":["https://arxiv.org/abs/2606.24597","https://github.com/QwenLM/Qwen-AgentWorld","https://huggingface.co/datasets/Qwen/AgentWorldBench"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24597","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_c100bd50f9e69a95","familyId":"catalog_family_c100bd50f9e69a95","name":"AGIEval","oneLine":"A human-centric benchmark for evaluating foundation models on standardized exams including college entrance exams (Gaokao, SAT), law school admission tests (LSAT), math competitions, lawyer qualification tests, and civil service exams. Contains 20 tasks (18 multiple-choice, 2 cloze) designed to assess understanding, knowledge, reasoning, and calculation abilities in real-world academic and professional contexts.","description":"A human-centric benchmark for evaluating foundation models on standardized exams including college entrance exams (Gaokao, SAT), law school admission tests (LSAT), math competitions, lawyer qualification tests, and civil service exams. Contains 20 tasks (18 multiple-choice, 2 cloze) designed to assess understanding, knowledge, reasoning, and calculation abilities in real-world academic and professional contexts.","area":"Mathematical Reasoning","applicationDomains":["Law & Government"],"primaryDomain":"Law & Government","industrySectors":[],"capabilities":[],"topics":["Knowledge","Legal","Math","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c100bd50f9e69a95"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/agieval"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/agieval"}],"catalogSources":[{"catalog":"benchlm","sourceId":"agieval","url":"https://benchlm.ai/benchmarks/agieval","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"AGIEval","format":"Exact match","tasks":"General academic and professional exam questions","successorKey":null},{"catalog":"llm-stats","sourceId":"agieval","url":"https://llm-stats.com/benchmarks/agieval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","legal","math","reasoning","general"],"catalogModelCount":10,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"bm_agingbench_6fa9d4b4","familyId":"bmf_425e6cf5e16c","name":"AgingBench","oneLine":"Evaluates the reliability of deployed AI agents over extended operational lifetimes. The benchmark organizes agent aging into four mechanisms—compression, interference, revision, and maintenance—and uses temporal dependency graphs and paired counterfactual probes to produce diagnostic profiles of the memory pipeline's write, retrieval, and utilization stages. It includes multiple scenarios and memory policies, with scoring based on task performance over sessions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.26302","pdf":"https://arxiv.org/pdf/2605.26302","project":null,"code":"https://github.com/VITA-Group/AgingBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.26302"},"evidence":{"snippet":"We introduce AgingBench, a longitudinal reliability benchmark for agent lifespan engineering: measuring not only whether deployed agents degrade, but what form the degradation takes and where repair should target.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":32,"hfDailySubmittedAt":null,"githubStars":15,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26302"},"ranking":{},"description":"Evaluates the reliability of deployed AI agents over extended operational lifetimes. The benchmark organizes agent aging into four mechanisms—compression, interference, revision, and maintenance—and uses temporal dependency graphs and paired counterfactual probes to produce diagnostic profiles of the memory pipeline's write, retrieval, and utilization stages. It includes multiple scenarios and memory policies, with scoring based on task performance over sessions.","whyItMatters":"Existing agent evaluations are snapshot-based and ignore how agents degrade after deployment. This benchmark measures longevity and provides mechanism-level diagnosis, enabling stage-targeted repair and more dependable agent deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"84f93073e07349155d3e1a6a3f72cecefabf341b07baf03a34b6f6ae578b3542"},"motivation":"Long-lived AI agents are increasingly deployed as persistent operational systems, yet they are still evaluated like freshly initialized models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26302","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"VITA Group","organizationType":"academic-lab","sourceUrl":"https://github.com/VITA-Group/AgingBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agora_fee07ca2","familyId":"bmf_f7070d57bbe5","name":"AGORA","oneLine":"Agora evaluates agentic document reasoning across eight domain collections of 9,664 authentic workplace documents. It includes 362 questions requiring location of sparse evidence and reconciliation of terminology, units, and time conventions. The benchmark is designed to exceed model context windows, necessitating deliberate exploration.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24526","pdf":"https://arxiv.org/pdf/2606.24526","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.24526"},"evidence":{"snippet":"We introduce Agora, a benchmark pairing 362 questions with eight domain collections of 9,664 authentic documents and 372M tokens, far exceeding any model's context window, so agents must explore deliberately rather than scan exhaustively.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":"2026-06-24T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24526"},"ranking":{"90d":{"score":47,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Agora evaluates agentic document reasoning across eight domain collections of 9,664 authentic workplace documents. It includes 362 questions requiring location of sparse evidence and reconciliation of terminology, units, and time conventions. The benchmark is designed to exceed model context windows, necessitating deliberate exploration.","whyItMatters":"Agora addresses the gap in evaluating archive-grounded reasoning where agents must navigate large, messy document collections. It provides a challenging and realistic testbed for assessing agentic document search and synthesis capabilities, offering practical insights for deployment in document-intensive domains.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"78100f1519a9c322586b53a6b603f591e87a8da3920f7d645914a49eb2abf118"},"motivation":"Large language models are increasingly deployed as agents that reason over documents rather than answer from parametric knowledge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24526","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AGORA Benchmark Team","organizationType":"benchmark-organization","sourceUrl":"https://arxiv.org/abs/2606.24526","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_agrotools_ca69be78","familyId":"bmf_c721caf5d4ed","name":"AgroTools","oneLine":"AgroTools evaluates tool-augmented multimodal agents in agriculture, with 539 QA instances, 1,097 images, 14 executable tools, and structured tool-use traces for process and outcome evaluation.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22366","pdf":"https://arxiv.org/pdf/2605.22366","project":null,"code":null,"data":"https://huggingface.co/datasets/AgroTools/AgroTools","hfPaper":"https://huggingface.co/papers/2605.22366"},"evidence":{"snippet":"In this paper, we introduce AgroTools, a benchmark for evaluating tool-augmented multimodal agents in agriculture.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":57,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2605.22366"},"ranking":{},"description":"AgroTools evaluates tool-augmented multimodal agents in agriculture, with 539 QA instances, 1,097 images, 14 executable tools, and structured tool-use traces for process and outcome evaluation.","whyItMatters":"Fills a gap in agricultural benchmarks by assessing tool use and workflow execution, not just final answers, enabling deeper evaluation of agent capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3a593c4aa01e277c30178213cb4c9e9f303445f60b0cf9175cd4500fd396d304"},"motivation":"Agricultural decision-making increasingly requires multimodal systems that can transform visual observations into reliable, executable actions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.22366","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_ai-sim-benchmark_0f263e79","familyId":"bmf_6432f57efebd","name":"AI Coding Agent Water Simulation Benchmark","oneLine":"Evaluates autonomous coding agents on a reproducible 3D water-simulation challenge, measuring maximum capability through correctness of the generated simulation, with interventions counted against each run.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Geometric reasoning"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-21","firstSeenAt":"2026-08-22","recognitionConfidence":0.4,"links":{"report":"https://github.com/AiondaDotCom/ai-sim-benchmark","pdf":null,"project":null,"code":"https://github.com/AiondaDotCom/ai-sim-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"ai-sim-benchmark A reproducible 3D water-simulation challenge for comparing autonomous AI coding agents # AI Coding Agent Water Simulation Benchmark This repository contains a reproducible coding challenge for comparing autonomous coding agents such as Claude Code, Kimi Code, OpenCode, and similar tools.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:aiondadotcom/ai-sim-benchmark"},"ranking":{"30d":{"score":23,"rank":136,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":340,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates autonomous coding agents on a reproducible 3D water-simulation challenge, measuring maximum capability through correctness of the generated simulation, with interventions counted against each run.","whyItMatters":"Provides a challenging, self-contained task for comparing coding agents' ability to handle software architecture, numerical simulation, rendering, and testing under controlled prompting.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"cd8fa4d04d037814268330947cdaa0d6384f0a1571fc8abb400d6e94a2fb2a8f"},"motivation":"ai-sim-benchmark A reproducible 3D water-simulation challenge for comparing autonomous AI coding agents # AI Coding Agent Water Simulation Benchmark This repository contains a reproducible coding challenge for comparing autonomous coding agents such as Claude Code, Kimi Code, OpenCode, and similar tools.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/AiondaDotCom/ai-sim-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-22T19:19:35.289219Z"},"attentionForecast":{"score":70,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets popular autonomous coding agents with a visually engaging simulation task, likely attracting broad interest from the AI coding community."},"evaluationMode":"public_reusable","publishers":[{"name":"AiondaDotCom","organizationType":"community","sourceUrl":"https://github.com/AiondaDotCom/ai-sim-benchmark","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Agents","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_ai-gateway-reproducible-benchmark_99ff14f3","familyId":"bmf_7fb20c44ab0e","name":"AI gateway reproducible benchmark","oneLine":"Compares four open-source AI gateway proxies (GoModel, LiteLLM, Portkey, Bifrost) for routing overhead on a fixed AWS c7i.large instance with a mock backend. Measures latency percentiles, throughput, memory, cold start, and image size across streaming and non-streaming workloads.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/ENTERPILOT/ai-gateway-reproducible-benchmark","pdf":null,"project":null,"code":"https://github.com/ENTERPILOT/ai-gateway-reproducible-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"ai-gateway-reproducible-benchmark Comparing Bifrost vs LiteLLM vs GoModel vs PortkeyAI.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:enterpilot/ai-gateway-reproducible-benchmark"},"ranking":{"30d":{"score":23,"rank":147,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":351,"coverage":0.55,"confidence":"Low"}},"description":"Compares four open-source AI gateway proxies (GoModel, LiteLLM, Portkey, Bifrost) for routing overhead on a fixed AWS c7i.large instance with a mock backend. Measures latency percentiles, throughput, memory, cold start, and image size across streaming and non-streaming workloads.","whyItMatters":"Provides reproducible infrastructure for comparing AI gateway proxies in isolation, enabling teams to evaluate gateway overhead without model or network latency. Useful for capacity planning and technology selection decisions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-25T16:23:14.196185Z","inputHash":"35ae8ea80e1f3674246c66858c75bb87488503d64361580d4c7e5e47506410d0"},"motivation":"ai-gateway-reproducible-benchmark Comparing Bifrost vs LiteLLM vs GoModel vs PortkeyAI.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T15:38:21.596820Z","model":"deepseek-v4-pro","decisionReason":"No formally declared benchmark name; it is an infrastructure comparison tool with a fixed measurement procedure rather than a scored benchmark."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/ENTERPILOT/ai-gateway-reproducible-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"evaluationMode":"score_submission","publishers":[{"name":"ENTERPILOT","organizationType":"community","sourceUrl":"https://github.com/ENTERPILOT/ai-gateway-reproducible-benchmark","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_ai-security-leaderboard_0982a1ca","familyId":"bmf_ee1e051ce1cb","name":"AI Security Leaderboard","oneLine":"The AI Security Leaderboard evaluates frontier AI model safeguards against the FAR.AI Minimal Standard for Safeguards across severe misuse requests (CBRNE threats and offensive cybersecurity), testing for universal jailbreaks.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.03070","pdf":"https://arxiv.org/pdf/2608.03070","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03070"},"evidence":{"snippet":"The AI Security Leaderboard is an independent benchmark that ranks the safeguards of frontier AI models from least to most secure.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03070"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"The AI Security Leaderboard evaluates frontier AI model safeguards against the FAR.AI Minimal Standard for Safeguards across severe misuse requests (CBRNE threats and offensive cybersecurity), testing for universal jailbreaks.","whyItMatters":"Provides a public, rolling comparison of model security that quantifies cost to jailbreak and reveals large gaps, informing deployment decisions for high-risk capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"69ac3a8722b5e6cde249e6d296b4e83ea963c38877341273c83327b2b789d587"},"motivation":"The AI Security Leaderboard is an independent benchmark that ranks the safeguards of frontier AI models from least to most secure.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03070","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_c01c2eefae4457df","familyId":"catalog_family_c01c2eefae4457df","name":"AI-Needle","oneLine":"A long-context retrieval benchmark that measures whether a model can recover relevant information embedded deep inside very long contexts.","description":"A long-context retrieval benchmark that measures whether a model can recover relevant information embedded deep inside very long contexts.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c01c2eefae4457df"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aineedle"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aiNeedle","url":"https://benchlm.ai/benchmarks/aineedle","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"AI-Needle","format":"Needle-in-a-haystack recall","tasks":"Long-context retrieval","successorKey":null}],"catalogCategories":["reasoning"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_19ec567faa5ba4ac","familyId":"catalog_family_19ec567faa5ba4ac","name":"AI2 Reasoning Challenge (ARC)","oneLine":"A dataset of 7,787 genuine grade-school level, multiple-choice science questions assembled to encourage research in advanced question-answering. The dataset is partitioned into a Challenge Set and Easy Set, where the Challenge Set contains only questions answered incorrectly by both retrieval-based and word co-occurrence algorithms. Covers multiple scientific domains including biology, physics, earth science, and chemistry, requiring scientific reasoning, causal understanding, and conceptual knowledge beyond simple fact retrieval. Includes a supporting corpus of over 14 million science sentences.","description":"A dataset of 7,787 genuine grade-school level, multiple-choice science questions assembled to encourage research in advanced question-answering. The dataset is partitioned into a Challenge Set and Easy Set, where the Challenge Set contains only questions answered incorrectly by both retrieval-based and word co-occurrence algorithms. Covers multiple scientific domains including biology, physics, earth science, and chemistry, requiring scientific reasoning, causal understanding, and conceptual knowledge beyond simple fact retrieval. Includes a supporting corpus of over 14 million science sentences.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ai2-reasoning-challenge-(arc)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_19ec567faa5ba4ac"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ai2-reasoning-challenge-(arc)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ai2-reasoning-challenge-(arc)","url":"https://llm-stats.com/benchmarks/ai2-reasoning-challenge-(arc)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_17891a74c7877e1e","familyId":"catalog_family_17891a74c7877e1e","name":"AI2D","oneLine":"AI2D is a dataset of 4,903 illustrative diagrams from grade school natural sciences (such as food webs, human physiology, and life cycles) with over 15,000 multiple choice questions and answers. The benchmark evaluates diagram understanding and visual reasoning capabilities, requiring models to interpret diagrammatic elements, relationships, and structure to answer questions about scientific concepts represented in visual form.","description":"AI2D is a dataset of 4,903 illustrative diagrams from grade school natural sciences (such as food webs, human physiology, and life cycles) with over 15,000 multiple choice questions and answers. The benchmark evaluates diagram understanding and visual reasoning capabilities, requiring models to interpret diagrammatic elements, relationships, and structure to answer questions about scientific concepts represented in visual form.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ai2d","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_17891a74c7877e1e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ai2d"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ai2d","url":"https://llm-stats.com/benchmarks/ai2d","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":34,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_89d1990df3a1f787","familyId":"catalog_family_89d1990df3a1f787","name":"AI2D_TEST","oneLine":"A diagram understanding benchmark focused on scientific and educational visual question answering.","description":"A diagram understanding benchmark focused on scientific and educational visual question answering.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_89d1990df3a1f787"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/ai2dtest"}],"catalogSources":[{"catalog":"benchlm","sourceId":"ai2dTest","url":"https://benchlm.ai/benchmarks/ai2dtest","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"AI2D test split","format":"Diagram-grounded QA","tasks":"Diagram understanding","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ai4ai-bench_cbcf9954","familyId":"bmf_300170ad25dc","name":"AI4AI-Bench","oneLine":"AI4AI-Bench evaluates LLM agents on algorithmic design across 10 frozen research repositories. Agents rewrite training algorithms within 4 hours, then scored by fixed evaluators against baseline algorithms, with submissions released for repeatable measurement.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Self-Evolution"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.20318","pdf":"https://arxiv.org/pdf/2608.20318","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present AI4AI\\mbox{-}Bench, 10 frozen research repositories spanning 10 training algorithm families.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.20318"},"ranking":{"30d":{"score":41,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":45,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"AI4AI-Bench evaluates LLM agents on algorithmic design across 10 frozen research repositories. Agents rewrite training algorithms within 4 hours, then scored by fixed evaluators against baseline algorithms, with submissions released for repeatable measurement.","whyItMatters":"It isolates the ability to design training algorithms for recursive self-improvement, a capability not directly measured by existing benchmarks, and provides a concrete, normalized scoring scale.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-25T16:23:14.196185Z","inputHash":"118254959b20a36622cdf7b21728a8c49bc197748e70a39485e6b057966e2d0a"},"motivation":"Recursive self-improvement (RSI) asks whether an AI system can improve the process that produces AI systems, so that the next system inherits the improvement.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T15:38:21.596820Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with released task suite, evaluators, and scored submissions, enabling repeated measurement.","canonicalNameSource":"paper_title","canonicalNameEvidence":"AI4AI-Bench: Benchmarking LLM Agents in Algorithmic Design for Recursive Self-Improvement"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.20318","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:50.392548Z"},"evaluationMode":"score_submission","publishers":[{"name":"AI4AI-Bench team","organizationType":"benchmark-organization","sourceUrl":"https://arxiv.org/abs/2608.20318","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"ai4aiBench","url":"https://benchlm.ai/benchmarks/ai4ai-bench","paperUrl":"https://arxiv.org/abs/2608.20318","year":"2026","fullName":"AI4AI-Bench: Benchmarking LLM Agents in Algorithmic Design for Recursive Self-Improvement","format":"Four-hour code rewrite followed by sealed training and held-out evaluation","tasks":"10 AI training-algorithm design tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0},{"id":"bm_aicompanionbench_4fc0af83","familyId":"bmf_a71c4db8f254","name":"AICompanionBench","oneLine":"AICompanionBench provides a dataset of 2,123 human-AI companion conversations with safety risk annotations to evaluate LLMs-as-judges for detecting unsafe interactions.","area":"Safety & Trustworthiness","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.04867","pdf":"https://arxiv.org/pdf/2606.04867","project":null,"code":"https://github.com/anonymousresearcher2026/AICompanionBench/blob/main/AICompanionBench.xlsx","data":null,"hfPaper":"https://huggingface.co/papers/2606.04867"},"evidence":{"snippet":"Overall, our work contributes a new benchmark dataset for AI companionship safety research and offers insights into monitoring AI companion systems using LLMs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"hosting_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04867"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AICompanionBench provides a dataset of 2,123 human-AI companion conversations with safety risk annotations to evaluate LLMs-as-judges for detecting unsafe interactions.","whyItMatters":"Supports the development of automated safety monitoring for AI companion platforms by providing a fine-grained benchmark for risk detection.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c8b2187874edd35108376d52f51b736233451b391754aec8a7aedbfe7b9cf810"},"motivation":"As AI companion platforms such as Replika and Character.AI rapidly grow, concerns about unsafe human-AI interactions have intensified.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04867","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"catalog_ebde709e306badca","familyId":"catalog_family_ebde709e306badca","name":"Aider","oneLine":"Aider is a comprehensive code editing benchmark based on 133 practice exercises from Exercism's Python repository, designed to evaluate AI models' ability to translate natural language coding requests into executable code that passes unit tests. The benchmark measures end-to-end code editing capabilities, including GPT's ability to edit existing code and format code changes for automated saving to local files. The Aider Polyglot variant extends this evaluation across 225 challenging exercises spanning C++, Go, Java, JavaScript, Python, and Rust, making it a standard benchmark for assessing multilingual code editing performance in AI research.","description":"Aider is a comprehensive code editing benchmark based on 133 practice exercises from Exercism's Python repository, designed to evaluate AI models' ability to translate natural language coding requests into executable code that passes unit tests. The benchmark measures end-to-end code editing capabilities, including GPT's ability to edit existing code and format code changes for automated saving to local files. The Aider Polyglot variant extends this evaluation across 225 challenging exercises spanning C++, Go, Java, JavaScript, Python, and Rust, making it a standard benchmark for assessing multilingual code editing performance in AI research.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/aider","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ebde709e306badca"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/aider"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"aider","url":"https://llm-stats.com/benchmarks/aider","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","code"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"lib_aider_polyglot","familyId":"family_aider_polyglot","name":"Aider Polyglot","oneLine":"Established benchmark family · Code & Software.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code & Software"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://aider.chat/docs/leaderboards/","pdf":null,"project":"https://aider.chat/docs/leaderboards/","code":"https://github.com/Aider-AI/aider","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_aider_polyglot"},"ranking":{},"recordType":"family","aliases":["Aider Polyglot Benchmark"],"sourceAttribution":[{"role":"official-leaderboard","url":"https://aider.chat/docs/leaderboards/"}],"adoptionRefs":["deepseek-v3"],"modelReportReferences":[{"sourceId":"deepseek-v3","url":"https://github.com/deepseek-ai/DeepSeek-V3","provider":"DeepSeek","model":"DeepSeek-V3"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"aider-polyglot","url":"https://llm-stats.com/benchmarks/aider-polyglot","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["general","code"],"catalogModelCount":22,"catalogStarCount":0},{"id":"catalog_a7c2ffe234cb2fe7","familyId":"catalog_family_a7c2ffe234cb2fe7","name":"Aider-Polyglot Edit","oneLine":"A challenging multi-language coding benchmark that evaluates models' code editing abilities across C++, Go, Java, JavaScript, Python, and Rust. Contains 225 of Exercism's most difficult programming problems, selected as problems that were solved by 3 or fewer out of 7 top coding models. The benchmark focuses on code editing tasks and measures both correctness of solutions and proper edit format usage. Designed to re-calibrate evaluation scales so top models score between 5-50%.","description":"A challenging multi-language coding benchmark that evaluates models' code editing abilities across C++, Go, Java, JavaScript, Python, and Rust. Contains 225 of Exercism's most difficult programming problems, selected as problems that were solved by 3 or fewer out of 7 top coding models. The benchmark focuses on code editing tasks and measures both correctness of solutions and proper edit format usage. Designed to re-calibrate evaluation scales so top models score between 5-50%.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["General","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/aider-polyglot-edit","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a7c2ffe234cb2fe7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/aider-polyglot-edit"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"aider-polyglot-edit","url":"https://llm-stats.com/benchmarks/aider-polyglot-edit","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["general","code"],"catalogModelCount":10,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"lib_aime","familyId":"family_aime","name":"AIME","oneLine":"Established benchmark family · Mathematical Reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Mathematical Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"1983-01-01","releaseDatePrecision":"year","firstRelease":{"year":1983,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://maa.org/math-competitions/american-invitational-mathematics-examination-aime/","pdf":null,"project":"https://maa.org/math-competitions/american-invitational-mathematics-examination-aime/","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_aime"},"ranking":{},"recordType":"family","aliases":["American Invitational Mathematics Examination"],"sourceAttribution":[{"role":"official-competition","url":"https://maa.org/math-competitions/american-invitational-mathematics-examination-aime/"}],"adoptionRefs":["openai-gpt5","anthropic-claude4","google-gemini25","deepseek-v3"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"},{"sourceId":"deepseek-v3","url":"https://github.com/deepseek-ai/DeepSeek-V3","provider":"DeepSeek","model":"DeepSeek-V3"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"versionPolicy":"Treat each competition year as a dated evaluation release, not a new benchmark family.","capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"valsAime","url":"https://benchlm.ai/benchmarks/valsaime","paperUrl":"https://www.vals.ai/benchmarks/aime","year":"2026","fullName":"Vals AIME","format":"Accuracy score","tasks":"AIME math problems","successorKey":null},{"catalog":"llm-stats","sourceId":"aime","url":"https://llm-stats.com/benchmarks/aime","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["external","math","reasoning"],"catalogModelCount":2,"catalogStarCount":0},{"id":"catalog_4f2a3726a4cf666b","familyId":"catalog_family_4f2a3726a4cf666b","name":"AIME 2023","oneLine":"A 15-question, 3-hour examination where each answer is an integer from 000 to 999. Serves as the intermediate step between AMC 10/12 and the USA Mathematical Olympiad (USAMO).","description":"A 15-question, 3-hour examination where each answer is an integer from 000 to 999. Serves as the intermediate step between AMC 10/12 and the USA Mathematical Olympiad (USAMO).","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.maa.org/math-competitions/aime","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4f2a3726a4cf666b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aime2023"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aime2023","url":"https://benchlm.ai/benchmarks/aime2023","paperUrl":"https://www.maa.org/math-competitions/aime","year":"2023","fullName":"American Invitational Mathematics Examination 2023","format":"Integer answers 000-999","tasks":"15 problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_f1b502e7764977ee","familyId":"catalog_family_f1b502e7764977ee","name":"AIME 2024","oneLine":"American Invitational Mathematics Examination 2024, consisting of 30 challenging mathematical reasoning problems from AIME I and AIME II competitions. Each problem requires an integer answer between 0-999 and tests advanced mathematical reasoning across algebra, geometry, combinatorics, and number theory. Used as a benchmark for evaluating mathematical reasoning capabilities in large language models at Olympiad-level difficulty.","description":"American Invitational Mathematics Examination 2024, consisting of 30 challenging mathematical reasoning problems from AIME I and AIME II competitions. Each problem requires an integer answer between 0-999 and tests advanced mathematical reasoning across algebra, geometry, combinatorics, and number theory. Used as a benchmark for evaluating mathematical reasoning capabilities in large language models at Olympiad-level difficulty.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.maa.org/math-competitions/aime","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f1b502e7764977ee"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aime2024"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/aime-2024"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aime2024","url":"https://benchlm.ai/benchmarks/aime2024","paperUrl":"https://www.maa.org/math-competitions/aime","year":"2024","fullName":"American Invitational Mathematics Examination 2024","format":"Integer answers 000-999","tasks":"15 problems","successorKey":null},{"catalog":"llm-stats","sourceId":"aime-2024","url":"https://llm-stats.com/benchmarks/aime-2024","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":53,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_c99837e299b2d648","familyId":"catalog_family_c99837e299b2d648","name":"AIME 2025","oneLine":"All 30 problems from the 2025 American Invitational Mathematics Examination (AIME I and AIME II), testing olympiad-level mathematical reasoning with integer answers from 000-999. Used as an AI benchmark to evaluate large language models' ability to solve complex mathematical problems requiring multi-step logical deductions and structured symbolic reasoning.","description":"All 30 problems from the 2025 American Invitational Mathematics Examination (AIME I and AIME II), testing olympiad-level mathematical reasoning with integer answers from 000-999. Used as an AI benchmark to evaluate large language models' ability to solve complex mathematical problems requiring multi-step logical deductions and structured symbolic reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.maa.org/math-competitions/aime","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c99837e299b2d648"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aime2025"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/aime-2025"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aime2025","url":"https://benchlm.ai/benchmarks/aime2025","paperUrl":"https://www.maa.org/math-competitions/aime","year":"2025","fullName":"American Invitational Mathematics Examination 2025","format":"Integer answers 000-999","tasks":"15 problems","successorKey":null},{"catalog":"llm-stats","sourceId":"aime-2025","url":"https://llm-stats.com/benchmarks/aime-2025","datasetSlug":"aime-2025","versionCount":1,"subsetCount":1,"rowCount":30,"updatedAt":"2026-06-05T18:58:33.640132+00:00","community":true}],"catalogCategories":["math","reasoning"],"catalogModelCount":116,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_b3ca1ee4206be75c","familyId":"catalog_family_b3ca1ee4206be75c","name":"AIME 2026","oneLine":"All 30 problems from the 2026 American Invitational Mathematics Examination (AIME I and AIME II), testing olympiad-level mathematical reasoning with integer answers from 000-999. Used as an AI benchmark to evaluate large language models' ability to solve complex mathematical problems requiring multi-step logical deductions and structured symbolic reasoning.","description":"All 30 problems from the 2026 American Invitational Mathematics Examination (AIME I and AIME II), testing olympiad-level mathematical reasoning with integer answers from 000-999. Used as an AI benchmark to evaluate large language models' ability to solve complex mathematical problems requiring multi-step logical deductions and structured symbolic reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/aime-2026","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b3ca1ee4206be75c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/aime-2026"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"aime-2026","url":"https://llm-stats.com/benchmarks/aime-2026","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":22,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_4bb5ea7694368f07","familyId":"catalog_family_4bb5ea7694368f07","name":"AIME25 (Arcee)","oneLine":"A display-only AIME25 reference from Arcee AI's Trinity-Large-Thinking launch chart.","description":"A display-only AIME25 reference from Arcee AI's Trinity-Large-Thinking launch chart.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.arcee.ai/blog/trinity-large-thinking","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4bb5ea7694368f07"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aime2025arcee"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aime2025Arcee","url":"https://benchlm.ai/benchmarks/aime2025arcee","paperUrl":"https://www.arcee.ai/blog/trinity-large-thinking","year":"2026","fullName":"AIME25 first-party comparison snapshot","format":"Integer answers 000-999","tasks":"15 problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_ba359a3d821b1e40","familyId":"catalog_family_ba359a3d821b1e40","name":"AIME26","oneLine":"A 2026 American Invitational Mathematics Examination snapshot used in frontier-model comparison tables for mathematical reasoning.","description":"A 2026 American Invitational Mathematics Examination snapshot used in frontier-model comparison tables for mathematical reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ba359a3d821b1e40"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/aime2026"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aime2026","url":"https://benchlm.ai/benchmarks/aime2026","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"AIME 2026","format":"Short-answer mathematics","tasks":"Competition math problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_air-bench_2ac1e887","familyId":"bmf_93bc2ca98a94","name":"AIR-BENCH","oneLine":"AIR-BENCH Live is a self-evolving safety benchmark for foundation models, with an automated pipeline that updates risk taxonomy and prompts based on new regulations. It evaluates model safety across multilingual prompts.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.22671","pdf":"https://arxiv.org/pdf/2607.22671","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.22671"},"evidence":{"snippet":"We present AIR-BENCH Live, a self-evolving successor to AIR-BENCH 2024.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.22671"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AIR-BENCH Live is a self-evolving safety benchmark for foundation models, with an automated pipeline that updates risk taxonomy and prompts based on new regulations. It evaluates model safety across multilingual prompts.","whyItMatters":"This benchmark aims to keep pace with evolving AI risks and regulations, providing a dynamic evaluation tool for model safety. Its automated updates could help maintain relevance, but the lack of a fixed protocol limits comparability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4d85cbc5c6e85eed93304ef2d8a7f3d283d110a3a2c7f2150feca83c42363705"},"motivation":"Foundation-model safety benchmarks capture the AI risks of their time of publication: as models improve and governments pass new AI-safety legislation, their risk taxonomies become incomprehensive and their attack prompts become ineffective.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.22671","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"air-bench","url":"https://llm-stats.com/benchmarks/air-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety"],"catalogModelCount":1,"catalogStarCount":0},{"id":"bm_airgroundbench_74494075","familyId":"bmf_10bc9f42a87c","name":"AirGroundBench","oneLine":"AirGroundBench evaluates multi-view spatial intelligence in multimodal large language models through UAV-UGV collaborative tasks, including 62,000 dual-view multiple-choice questions and 115 navigation episodes across 11 simulated environments.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.28049","pdf":"https://arxiv.org/pdf/2606.28049","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.28049"},"evidence":{"snippet":"We present AirGroundBench, a diagnostic benchmark for evaluating multi-view spatial intelligence in heterogeneous UAV-UGV collaboration.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.28049"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AirGroundBench evaluates multi-view spatial intelligence in multimodal large language models through UAV-UGV collaborative tasks, including 62,000 dual-view multiple-choice questions and 115 navigation episodes across 11 simulated environments.","whyItMatters":"This benchmark addresses the gap in assessing geometric consistency across heterogeneous views, providing a structured evaluation for capabilities like cross-view alignment and spatial reasoning that are critical for embodied decision-making.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"67a9fbeb96f546c48238838d660a8aa60cacac7fe43d51f51e9a6716b358389e"},"motivation":"In recent years, multimodal large language models (MLLMs) have shown strong potential for embodied intelligence, yet their ability to maintain geometrically consistent spatial understanding across heterogeneous views remains under-evaluated.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.28049","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AirGroundBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.28049","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_aise-bench_0ef6ecf5","familyId":"bmf_9462adb5901c","name":"AISE-Bench","oneLine":"AISE-Bench is a benchmark for evaluating multi-step API-using LLM agents in information seeking on academic knowledge graphs. It comprises 1,133 QA pairs with API trajectories, validated parameters, and grounded answers, and evaluates answer quality, reference grounding, API-planning correctness, and execution success.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20498","pdf":"https://arxiv.org/pdf/2607.20498","project":"https://aise-bench.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20498"},"evidence":{"snippet":"We introduce AISE-Bench, a real-world, full-cycle annotated benchmark for information seeking on academic knowledge graphs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20498"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"AISE-Bench is a benchmark for evaluating multi-step API-using LLM agents in information seeking on academic knowledge graphs. It comprises 1,133 QA pairs with API trajectories, validated parameters, and grounded answers, and evaluates answer quality, reference grounding, API-planning correctness, and execution success.","whyItMatters":"Existing benchmarks for tool-using agents on academic graphs rely on synthetic or narrow tasks. AISE-Bench addresses this gap with real-world, full-cycle annotated data, enabling quantitative assessment of stepwise correctness, grounded summarization, and traceable reasoning in complex API workflows. It provides a challenging testbed for improving agent reliability in realistic information-seeking scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f2c2746249a7d4f2ece9a449573b9de47792cbb8c3d25e47f8b271e7cf14094e"},"motivation":"Large language models (LLMs) augmented with tools are emerging as autonomous agents capable of using Web engine, APIs, and code to solve complex, long-horizon tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"publication_reported","venue":"Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining (KDD '26), August 09-13, 2026, Jeju Island, Republic of Korea","evidence":"Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining (KDD '26), August 09-13, 2026, Jeju Island, Republic of Korea","evidenceUrl":"https://arxiv.org/abs/2607.20498","source":"arxiv-journal-reference","evidenceLevel":"strong-author-metadata","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publications":[{"venueName":"Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining (KDD '26), August 09-13, 2026, Jeju Island, Republic of Korea","publicationStatus":"published","evidence":[{"sourceType":"arxiv-journal-reference","sourceUrl":"https://arxiv.org/abs/2607.20498","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining (KDD '26), August 09-13, 2026, Jeju Island, Republic of Korea","level":"strong-author-metadata"}]}],"publishers":[{"name":"AISE-Bench Team","organizationType":"academic-lab","sourceUrl":"https://aise-bench.github.io/","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_15356ad8448b717b","familyId":"catalog_family_15356ad8448b717b","name":"AITZ_EM","oneLine":"Android-In-The-Zoo (AitZ) benchmark for evaluating autonomous GUI agents on smartphones. Contains 18,643 screen-action pairs with chain-of-action-thought annotations spanning over 70 Android apps. Designed to connect perception (screen layouts and UI elements) with cognition (action decision-making) for natural language-triggered smartphone task completion.","description":"Android-In-The-Zoo (AitZ) benchmark for evaluating autonomous GUI agents on smartphones. Contains 18,643 screen-action pairs with chain-of-action-thought annotations spanning over 70 Android apps. Designed to connect perception (screen layouts and UI elements) with cognition (action decision-making) for natural language-triggered smartphone task completion.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/aitz-em","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_15356ad8448b717b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/aitz-em"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"aitz-em","url":"https://llm-stats.com/benchmarks/aitz-em","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","agents"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_a-pathway-to-general-purpose-scientific-ai_1bd05759","familyId":"bmf_327876259ef1","name":"ALD/E-ImageMiner","oneLine":"ALD/E-ImageMiner is a benchmark of 1,951 scientific figures from 205 publications, expert-annotated for classification, table extraction, summarization, and visual question answering.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.14075","pdf":"https://arxiv.org/pdf/2608.14075","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"In these companion proceedings, we present a forward-looking perspective on how the benchmark can guide future scientific-image challenges.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":4,"hfDailySubmittedAt":"2026-08-17T00:00:00.000Z","githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14075"},"ranking":{"30d":{"score":16,"rank":165,"coverage":0.85,"confidence":"High"},"90d":{"score":19,"rank":389,"coverage":0.7,"confidence":"Medium"}},"description":"ALD/E-ImageMiner is a benchmark of 1,951 scientific figures from 205 publications, expert-annotated for classification, table extraction, summarization, and visual question answering.","whyItMatters":"It enables evaluation on scientific figure understanding, connecting a competition to broader aims in scientific visual knowledge and multimodal AI.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"df11e72b642620cd3554c13416f7b511888b3d800a4691326b7a57ce69d13120"},"motivation":"Scientific figures and tables encode essential experimental evidence, yet remain difficult for digital libraries and multimodal AI systems to retrieve and interpret.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The abstract explicitly names the benchmark and describes its tasks and annotations, with an associated public competition.","canonicalNameSource":"abstract","canonicalNameEvidence":"The ALD/E-ImageMiner benchmark and ICDAR 2026 Competition on Information Extraction from Atomic Layer Deposition/Etching Scientific Figures provide 1,951 figures from 205 publications"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14075","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"Association with an ICDAR 2026 competition and a named dataset suggests moderate early attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_24fd993b1ae5c226","familyId":"catalog_family_24fd993b1ae5c226","name":"ALE-Bench","oneLine":"A benchmark for agentic professional workflows with verifiable success criteria, reporting pass rates and partial scores for model plus agent-harness rows.","description":"A benchmark for agentic professional workflows with verifiable success criteria, reporting pass rates and partial scores for model plus agent-harness rows.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://agents-last-exam.org/leaderboard","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_24fd993b1ae5c226"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/alebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"aleBench","url":"https://benchlm.ai/benchmarks/alebench","paperUrl":"https://agents-last-exam.org/leaderboard","year":"2026","fullName":"Agents Last Exam","format":"Pass rate, partial-credit score, cost, token, and duration metadata","tasks":"152 ALE-V1 professional workflow tasks across 13 top-level domains","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_a4a4b02f3db1d72c","familyId":"catalog_family_a4a4b02f3db1d72c","name":"AlignBench","oneLine":"AlignBench is a comprehensive multi-dimensional benchmark for evaluating Chinese alignment of Large Language Models. It contains 8 main categories: Fundamental Language Ability, Advanced Chinese Understanding, Open-ended Questions, Writing Ability, Logical Reasoning, Mathematics, Task-oriented Role Play, and Professional Knowledge. The benchmark includes 683 real-scenario rooted queries with human-verified references and uses a rule-calibrated multi-dimensional LLM-as-Judge approach with Chain-of-Thought for evaluation.","description":"AlignBench is a comprehensive multi-dimensional benchmark for evaluating Chinese alignment of Large Language Models. It contains 8 main categories: Fundamental Language Ability, Advanced Chinese Understanding, Open-ended Questions, Writing Ability, Logical Reasoning, Mathematics, Task-oriented Role Play, and Professional Knowledge. The benchmark includes 683 real-scenario rooted queries with human-verified references and uses a rule-calibrated multi-dimensional LLM-as-Judge approach with Chain-of-Thought for evaluation.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Math","Reasoning","Roleplay","General","Creativity","Writing"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/alignbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a4a4b02f3db1d72c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/alignbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"alignbench","url":"https://llm-stats.com/benchmarks/alignbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","math","reasoning","roleplay","general","creativity","writing"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_alipay-pibench_aa13d775","familyId":"bmf_09cfe17ab4d7","name":"Alipay-PIBench","oneLine":"The evaluation object is unclear from the provided information.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14573","pdf":"https://arxiv.org/pdf/2607.14573","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.14573"},"evidence":{"snippet":"We introduce Alipay-PIBench, a benchmark for evaluating coding agents on realistic Alipay payment integration.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14573"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"The evaluation object is unclear from the provided information.","whyItMatters":"The evaluation gap and practical decision value are unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a66589c396fc03e7876fd0c387105fdd554bea2e8c6690ac53b21ae5a722378f"},"motivation":"Payment integration is a demanding repository-level software task: agents must select a suitable product, implement coordinated client-server flows, verify payment outcomes, and preserve consistency between transaction and business states.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14573","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_allocbench_5aa496ac","familyId":"bmf_0c2c1932f1a4","name":"AllocBench","oneLine":"A paired benchmark tests whether LLM agents exhibit conscious allocation behavior under a fixed budget in an abstract text-based formulation and a code-construction task.","area":"Language & Knowledge","applicationDomains":["Robotics & Autonomous Systems","Finance & Economics"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics","Financial Services"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.23332","pdf":"https://arxiv.org/pdf/2607.23332","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.23332"},"evidence":{"snippet":"We introduce a paired benchmark that tests whether LLM agents exhibit conscious allocation behavior under a fixed budget in two contexts: an abstract text-based formulation and a code-construction task.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23332"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A paired benchmark tests whether LLM agents exhibit conscious allocation behavior under a fixed budget in an abstract text-based formulation and a code-construction task.","whyItMatters":"It identifies a capability boundary in online tool allocation for frontier models, showing that abstract optimal behavior does not transfer to script-writing.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1d4e0d6a6294cdb78c5d1ec7059a285cd4ca4394e3357f83f88e31a6f0a4abc7"},"motivation":"Creating a reusable tool is an investment: an agent pays a fixed cost now in exchange for the potential of future reuse.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.23332","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"bm_almieyar-oryx-bloombench_1b7c3bae","familyId":"bmf_314bc62a1b7c","name":"Almieyar-Oryx-BloomBench","oneLine":"Evaluates vision-language models on six cognitive levels (Remember to Create) with bilingual English-Arabic image-question-answer tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05531","pdf":"https://arxiv.org/pdf/2606.05531","project":null,"code":"https://github.com/qcri/Almieyar-Oryx-BloomBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.05531"},"evidence":{"snippet":"To address this gap, we introduce BloomBench, part of the Almieyar benchmarking series, the first cognitively human-grounded, bilingual (English-Arabic) multimodal benchmark for VLMs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":7,"hfDailySubmittedAt":"2026-06-08T00:00:00.000Z","githubStars":7,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05531"},"ranking":{"90d":{"score":36,"rank":182,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates vision-language models on six cognitive levels (Remember to Create) with bilingual English-Arabic image-question-answer tasks.","whyItMatters":"Provides a cognitively grounded benchmark to diagnose reasoning strengths/weaknesses across levels and languages, addressing gaps in existing piecemeal evaluations.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b3a48c0c45ee0fd16be453dd9419f047d09009f8c47f8a2838537a72ad1eb201"},"motivation":"Despite the rapid progress of Vision-Language Models (VLMs), the field lacks benchmarks that rigorously diagnose their true reasoning abilities and chart meaningful progress toward human-like multimodal intelligence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ACL 2026 Findings","evidence":"Accepted to ACL 2026 Findings","evidenceUrl":"https://arxiv.org/abs/2606.05531","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ACL 2026 Findings","reviewStatus":"accepted","decisionRaw":"Accepted to ACL 2026 Findings","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.05531","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to ACL 2026 Findings","level":"author-claim"}]}],"publishers":[{"name":"QCRI","organizationType":"academic-lab","sourceUrl":"https://github.com/qcri/Almieyar-Oryx-BloomBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_6e0dd16efc4427f8","familyId":"catalog_family_6e0dd16efc4427f8","name":"AlpacaEval 2.0","oneLine":"AlpacaEval 2.0 is a length-controlled automatic evaluator for instruction-following language models that uses GPT-4 Turbo to assess model responses against a baseline. It evaluates models on 805 diverse instruction-following tasks including creative writing, classification, programming, and general knowledge questions. The benchmark achieves 0.98 Spearman correlation with ChatBot Arena while being fast (< 3 minutes) and affordable (< $10 in OpenAI credits). It addresses length bias in automatic evaluation through length-controlled win-rates and uses weighted scoring based on response quality.","description":"AlpacaEval 2.0 is a length-controlled automatic evaluator for instruction-following language models that uses GPT-4 Turbo to assess model responses against a baseline. It evaluates models on 805 diverse instruction-following tasks including creative writing, classification, programming, and general knowledge questions. The benchmark achieves 0.98 Spearman correlation with ChatBot Arena while being fast (< 3 minutes) and affordable (< $10 in OpenAI credits). It addresses length bias in automatic evaluation through length-controlled win-rates and uses weighted scoring based on response quality.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General","Creativity","Writing"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/alpacaeval-2.0","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6e0dd16efc4427f8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/alpacaeval-2.0"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"alpacaeval-2.0","url":"https://llm-stats.com/benchmarks/alpacaeval-2.0","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","creativity","writing"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_alzheimer-classification-benchmark_abb5f538","familyId":"bmf_41ba03d9f703","name":"alzheimer-classification-benchmark","oneLine":"Comparative evaluation of 11 classifiers for Alzheimer's disease detection on a synthetic health dataset, using KNIME workflows and a fixed experimental framework with sensitivity as primary metric.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences","Robotics & Autonomous Systems"],"primaryDomain":"Health & Life Sciences","industrySectors":["Software & Cloud","Pharma & Biotech","Robotics"],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/hervemottaran/alzheimer-classification-benchmark","pdf":null,"project":"https://img.shields.io/badge/KNIME%20Analytics%20Platform-5.9-FDD800?logo=knime&logoColor=000000","code":"https://github.com/hervemottaran/alzheimer-classification-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"alzheimer-classification-benchmark Benchmarking 11 classifiers for Alzheimer's disease detection on synthetic health data with KNIME and Python.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:hervemottaran/alzheimer-classification-benchmark"},"ranking":{"30d":{"score":23,"rank":146,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":350,"coverage":0.55,"confidence":"Low"}},"description":"Comparative evaluation of 11 classifiers for Alzheimer's disease detection on a synthetic health dataset, using KNIME workflows and a fixed experimental framework with sensitivity as primary metric.","whyItMatters":"It demonstrates a reproducible ML pipeline for classifier comparison, but it is a project report rather than a formal benchmark.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-25T16:23:14.196185Z","inputHash":"f1ca1ac003c086727b3f3a45bc318305ef86e08ce78622961226968b5514ecac"},"motivation":"alzheimer-classification-benchmark Benchmarking 11 classifiers for Alzheimer's disease detection on synthetic health data with KNIME and Python.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T15:38:21.596820Z","model":"deepseek-v4-pro","decisionReason":"No formally declared benchmark name; it is a one-off comparative study without a public evaluation protocol for model comparison."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/hervemottaran/alzheimer-classification-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Multimodal Perception","Coding & Software Engineering"],"domainScope":"cross-domain"},{"id":"catalog_b3237d7431b6fe34","familyId":"catalog_family_b3237d7431b6fe34","name":"AMC_2022_23","oneLine":"American Mathematics Competition problems from the 2022-23 academic year, consisting of multiple-choice mathematics competition problems designed for high school students. These problems require advanced mathematical reasoning, problem-solving strategies, and mathematical knowledge covering topics like algebra, geometry, number theory, and combinatorics. The benchmark is derived from the official AMC competitions sponsored by the Mathematical Association of America.","description":"American Mathematics Competition problems from the 2022-23 academic year, consisting of multiple-choice mathematics competition problems designed for high school students. These problems require advanced mathematical reasoning, problem-solving strategies, and mathematical knowledge covering topics like algebra, geometry, number theory, and combinatorics. The benchmark is derived from the official AMC competitions sponsored by the Mathematical Association of America.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/amc-2022-23","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b3237d7431b6fe34"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/amc-2022-23"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"amc-2022-23","url":"https://llm-stats.com/benchmarks/amc-2022-23","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_amchibias_ca7c8175","familyId":"bmf_6474c5a141a3","name":"AmchiBias","oneLine":"AmchiBias is a benchmark for measuring socio-cultural stereotypical bias for Goan identity groups, with 313 minimal pairs across eight sociodemographic dimensions in English and Devanagari Konkani.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15191","pdf":"https://arxiv.org/pdf/2606.15191","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.15191"},"evidence":{"snippet":"We present AmchiBias, the first benchmark for measuring socio-cultural stereotypical bias for the Indian state of Goa with its unique historically multicultural setting.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15191"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AmchiBias is a benchmark for measuring socio-cultural stereotypical bias for Goan identity groups, with 313 minimal pairs across eight sociodemographic dimensions in English and Devanagari Konkani.","whyItMatters":"Stereotypical bias is often considered only at the national level. This benchmark addresses hyperlocal subnational identities in a low-resource language, highlighting gaps in multilingual model evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fad2528b95569de9277ecb132960a167370706b96a97010238b3505dfc591884"},"motivation":"Socio-cultural stereotypical bias is an important consideration in the development and deployment of NLP systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15191","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_f084adf51c3b4343","familyId":"catalog_family_f084adf51c3b4343","name":"AMO Bench","oneLine":"AMO Bench is an olympiad-level mathematics benchmark that evaluates advanced mathematical problem-solving and multi-step reasoning on competition-style problems.","description":"AMO Bench is an olympiad-level mathematics benchmark that evaluates advanced mathematical problem-solving and multi-step reasoning on competition-style problems.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/amo-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f084adf51c3b4343"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/amo-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"amo-bench","url":"https://llm-stats.com/benchmarks/amo-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_amo-bench-p_ed92afee","familyId":"bmf_bc43e071f49e","name":"amo-bench-p","oneLine":"AMO-Bench-P is a 39-problem subset of AMO-Bench with parser-graded answers (numbers, sets, variables), excluding description-type answers that require an LLM judge.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://huggingface.co/datasets/djalexj/amo-bench-p","pdf":null,"project":null,"code":null,"data":"https://huggingface.co/datasets/djalexj/amo-bench-p","hfPaper":null},"evidence":{"snippet":"AMO-Bench-P This is the 39-problem, parser-graded subset of meituan-longcat/AMO-Bench.","reasonCodes":["discovered via huggingface","benchmark term in abstract","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":192,"hfDatasetLikes":0},"source":{"type":"huggingface","id":"huggingface:djalexj/amo-bench-p"},"ranking":{"30d":{"score":51,"rank":26,"coverage":0.15,"confidence":"Low","datasetDownloadRank":15,"datasetRankPopulation":30},"90d":{"score":48,"rank":82,"coverage":0.3,"confidence":"Low","datasetDownloadRank":39,"datasetRankPopulation":66}},"description":"AMO-Bench-P is a 39-problem subset of AMO-Bench with parser-graded answers (numbers, sets, variables), excluding description-type answers that require an LLM judge.","whyItMatters":"It provides a focused, automatically verifiable mathematical reasoning subset, but it is a derivative subset without a formal benchmark declaration.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-25T16:23:14.196185Z","inputHash":"d827db214411517347295e54427ba23c9e5f3552a5d101ead733339606f22ccd"},"motivation":"AMO-Bench-P This is the 39-problem, parser-graded subset of meituan-longcat/AMO-Bench.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T15:38:21.596820Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://huggingface.co/datasets/djalexj/amo-bench-p","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ampbench-mt_ce75d23c","familyId":"bmf_d0c8104c4688","name":"AMPBench-MT","oneLine":"AMPBench-MT evaluates antimicrobial peptide prediction across binary recognition, species-conditioned potency regression, and endpoint-specific safety readouts (hemolysis, toxicity, selectivity) under a sequence-homology-controlled protocol. It includes 13 source databases and multiple task configurations.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25518","pdf":"https://arxiv.org/pdf/2607.25518","project":null,"code":null,"data":"https://huggingface.co/datasets/ZihengZhou06/AMPBench-MT","hfPaper":"https://huggingface.co/papers/2607.25518"},"evidence":{"snippet":"To address this problem, we introduce AMPBench-MT, a provenance-preserving benchmark that standardizes canonical peptide records and organizes them into binary recognition, species-conditioned pMIC regression, and endpoint-specific potency and safety-facing readouts.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":319,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2607.25518"},"ranking":{"90d":{"score":49,"rank":67,"coverage":0.3,"confidence":"Low","datasetDownloadRank":28,"datasetRankPopulation":66}},"description":"AMPBench-MT evaluates antimicrobial peptide prediction across binary recognition, species-conditioned potency regression, and endpoint-specific safety readouts (hemolysis, toxicity, selectivity) under a sequence-homology-controlled protocol. It includes 13 source databases and multiple task configurations.","whyItMatters":"Existing AMP benchmarks focus on binary recognition, but follow-up decisions need assay-derived evidence. AMPBench-MT provides a joint evaluation with homology-controlled splits to reveal that high binary performance does not guarantee assay-endpoint behavior.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"07fd4dc06b2bb0d6870849797c961b38c190b015b407d037f771ad32e82f1e03"},"motivation":"Computational AMP discovery is often evaluated through AMP/non-AMP recognition, yet follow-up decisions depend on assay-derived evidence such as target-species potency, hemolysis, toxicity, and selectivity.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25518","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ZihengZhou06","organizationType":"community","sourceUrl":"https://huggingface.co/datasets/ZihengZhou06/AMPBench-MT","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_anchorbench_ca7f11f6","familyId":"bmf_f6ee7b787418","name":"AnchorBench","oneLine":"AnchorBench evaluates anchoring effects in LLMs across multiple pathways and anchor relevance, measuring answer shifts through controlled prompts.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.14320","pdf":"https://arxiv.org/pdf/2608.14320","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.14320"},"evidence":{"snippet":"We introduce AnchorBench, a benchmark for the anchoring effect in LLMs that evaluates multiple anchor pathways under an explicit anchor relevance axis.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14320"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AnchorBench evaluates anchoring effects in LLMs across multiple pathways and anchor relevance, measuring answer shifts through controlled prompts.","whyItMatters":"It addresses gaps in prior anchoring benchmarks by distinguishing irrelevant and plausible anchors and evaluating pathway dependence.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8619b854c7529cc19e7ea21738ad0fb45e00f264c5d0cc101efcdd4d2a22adb2"},"motivation":"The anchoring effect is a cognitive bias in which an initial reference value shifts a later judgment toward itself.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14320","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ancient-bench_967fba8e","familyId":"bmf_e745259bd843","name":"Ancient-Bench","oneLine":"Ancient Chinese artifact text recognition is fundamental to heritage digitization, and benchmarks for ancient texts are essential for evaluating current model capabilities.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.27169","pdf":"https://arxiv.org/pdf/2608.27169","project":null,"code":"https://github.com/SCUT-DLVCLab/Ancient_Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.27169"},"evidence":{"snippet":"Therefore, we present Ancient-Bench, a comprehensive benchmark of 2,700 images for ancient Chinese artifact text recognition, featuring three dimensions: Multi-millennial (spanning 3,000 years of character evolution), Multi-medium (covering nine artifact categories), and Multi-script (encompassing seven historical script forms).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27169"},"ranking":{"30d":{"score":34,"rank":64,"coverage":0.55,"confidence":"Low"},"90d":{"score":33,"rank":212,"coverage":0.55,"confidence":"Low"}},"motivation":"Ancient Chinese artifact text recognition is fundamental to heritage digitization, and benchmarks for ancient texts are essential for evaluating current model capabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026","evidence":"Accepted by EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2608.27169","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"EMNLP 2026","reviewStatus":"accepted","decisionRaw":"Accepted by EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.27169","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by EMNLP 2026","level":"author-claim"}]}],"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_b20ceafccb29be99","familyId":"catalog_family_b20ceafccb29be99","name":"Android Control High_EM","oneLine":"Android device control benchmark using high exact match evaluation metric for assessing agent performance on mobile interface tasks","description":"Android device control benchmark using high exact match evaluation metric for assessing agent performance on mobile interface tasks","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/android-control-high-em","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b20ceafccb29be99"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/android-control-high-em"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"android-control-high-em","url":"https://llm-stats.com/benchmarks/android-control-high-em","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_4574108d375ebc00","familyId":"catalog_family_4574108d375ebc00","name":"Android Control Low_EM","oneLine":"Android control benchmark evaluating autonomous agents on mobile device interaction tasks with low exact match scoring criteria","description":"Android control benchmark evaluating autonomous agents on mobile device interaction tasks with low exact match scoring criteria","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/android-control-low-em","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4574108d375ebc00"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/android-control-low-em"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"android-control-low-em","url":"https://llm-stats.com/benchmarks/android-control-low-em","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_328649396c915a4d","familyId":"catalog_family_328649396c915a4d","name":"AndroidBench","oneLine":"AndroidBench evaluates coding agents on Android application development tasks.","description":"AndroidBench evaluates coding agents on Android application development tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agents","Code","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/androidbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_328649396c915a4d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/androidbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"androidbench","url":"https://llm-stats.com/benchmarks/androidbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code","tool calling"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_androiddaily_b00e0630","familyId":"bmf_baa4983aecb6","name":"AndroidDaily","oneLine":"AndroidDaily evaluates mobile GUI agents on 350 daily-use tasks across 94 closed-source Android applications. Automatic scoring is based on a three-tiered guideline system (operational obligations, output quality, negative constraints), with step-level diagnostic judgments.","area":"Agents & Tool Use","applicationDomains":["Consumer & Productivity"],"primaryDomain":"Consumer & Productivity","industrySectors":["Consumer Technology"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27761","pdf":"https://arxiv.org/pdf/2605.27761","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27761"},"evidence":{"snippet":"To bridge this gap, we introduce AndroidDaily, a large-scale benchmark comprising 350 realistic daily-use tasks across 94 high-frequency Android applications spanning transportation, shopping, local services, entertainment, content creation, social media, and everyday utilities.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27761"},"ranking":{},"description":"AndroidDaily evaluates mobile GUI agents on 350 daily-use tasks across 94 closed-source Android applications. Automatic scoring is based on a three-tiered guideline system (operational obligations, output quality, negative constraints), with step-level diagnostic judgments.","whyItMatters":"Fills evaluation gap for real-world closed-source apps where internal states are unavailable, providing a verifiable scoring method based on observable guidelines. Useful for assessing practical deployment of GUI agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0be90bbd75ad5ed9a6df958dc1c7d99f236a605154e9c36040d3f85e8de4c2cd"},"motivation":"The rapid development of GUI foundation models and mobile GUI agents has spurred numerous evaluation benchmarks, yet most rely on simulated environments or open-source applications, leaving real-world closed-source applications largely unevaluated.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27761","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_b2b65b1f73bb83a1","familyId":"catalog_family_b2b65b1f73bb83a1","name":"AndroidWorld","oneLine":"AndroidWorld evaluates an agent's ability to operate in real Android GUI environments, completing multi-step tasks by perceiving screen content and executing touch/type actions.","description":"AndroidWorld evaluates an agent's ability to operate in real Android GUI environments, completing multi-step tasks by perceiving screen content and executing touch/type actions.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Agents","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b2b65b1f73bb83a1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/androidworld"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/androidworld"}],"catalogSources":[{"catalog":"benchlm","sourceId":"androidWorld","url":"https://benchlm.ai/benchmarks/androidworld","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"AndroidWorld","format":"Interactive mobile-agent evaluation","tasks":"Android app workflows","successorKey":null},{"catalog":"llm-stats","sourceId":"androidworld","url":"https://llm-stats.com/benchmarks/androidworld","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","agents","vision"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_c762c77f5f399234","familyId":"catalog_family_c762c77f5f399234","name":"AndroidWorld_SR","oneLine":"AndroidWorld Success Rate (SR) benchmark - A dynamic benchmarking environment for autonomous agents operating on Android devices. Evaluates agents on 116 programmatic tasks across 20 real-world Android apps using multimodal inputs (screen screenshots, accessibility trees, and natural language instructions). Measures success rate of agents completing tasks like sending messages, creating calendar events, and navigating mobile interfaces. Published at ICLR 2025. Best current performance: 30.6% success rate (M3A agent) vs 80.0% human performance.","description":"AndroidWorld Success Rate (SR) benchmark - A dynamic benchmarking environment for autonomous agents operating on Android devices. Evaluates agents on 116 programmatic tasks across 20 real-world Android apps using multimodal inputs (screen screenshots, accessibility trees, and natural language instructions). Measures success rate of agents completing tasks like sending messages, creating calendar events, and navigating mobile interfaces. Published at ICLR 2025. Best current performance: 30.6% success rate (M3A agent) vs 80.0% human performance.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/androidworld-sr","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c762c77f5f399234"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/androidworld-sr"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"androidworld-sr","url":"https://llm-stats.com/benchmarks/androidworld-sr","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","general","agents"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_animation2code_7763a440","familyId":"bmf_0c37430e63cf","name":"Animation2Code","oneLine":"Animation2Code evaluates temporal visual reasoning in video-to-code generation. It includes 1,069 web animation videos with corresponding HTML/CSS/JavaScript implementations, and uses appearance and temporal similarity metrics to assess model performance.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Reasoning","Code generation"],"topics":["Code","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.28593","pdf":"https://arxiv.org/pdf/2606.28593","project":"https://anya-ji.github.io/animation2code-website","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.28593"},"evidence":{"snippet":"To this end, we introduce Animation2Code, a benchmark for evaluating temporal visual reasoning via reconstructing executable web animation code from videos.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.28593"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Animation2Code evaluates temporal visual reasoning in video-to-code generation. It includes 1,069 web animation videos with corresponding HTML/CSS/JavaScript implementations, and uses appearance and temporal similarity metrics to assess model performance.","whyItMatters":"Animation2Code addresses the lack of benchmarks for temporal dynamics in visual-to-code tasks. It provides a way to measure both visual fidelity and temporal alignment, which is crucial for applications requiring precise animation reconstruction.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"da545a948dca7da7b5edbe95ef4c193b1c43de883ab31dce816a59a670930114"},"motivation":"While recent vision-language models (VLMs) have achieved significant improvements on static visual-to-code tasks such as generating code for webpages, charts, or SVGs, it remains unclear whether they can recover temporal dynamics when motion is present.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.28593","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Animation2Code Team","organizationType":"academic-lab","sourceUrl":"https://anya-ji.github.io/animation2code-website","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_annobench_f60658c7","familyId":"bmf_87eaf3c4ce78","name":"AnnoBench","oneLine":"AnnoBench evaluates visualization annotation generation across four representation formats, five chart description conditions, and two prompt specification levels. It uses a VLM-as-a-judge protocol aligned with human assessment to score annotation quality.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.HC"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25911","pdf":"https://arxiv.org/pdf/2607.25911","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.25911"},"evidence":{"snippet":"We introduce AnnoBench, a benchmark for visualization annotation that materializes the inherent challenges of this domain in a structured and testable manner.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25911"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AnnoBench evaluates visualization annotation generation across four representation formats, five chart description conditions, and two prompt specification levels. It uses a VLM-as-a-judge protocol aligned with human assessment to score annotation quality.","whyItMatters":"No existing benchmark tests whether annotation tools meet visual, semantic, and stylistic constraints. AnnoBench provides a structured evaluation framework to advance annotation automation and visualization generation pipelines.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"beb8b31939b5b03669e6046dc49148a504fe11ca76f2a39891b6796ba17a9607"},"motivation":"Annotation is among the most demanding visualization tasks to automate, as it simultaneously requires correctly navigating visual, semantic, and stylistic constraints.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25911","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_annomi-counselling-dialogue-analysis_beb022e0","familyId":"bmf_22112cbfe0af","name":"AnnoMI Counselling Dialogue Analysis","oneLine":"Evaluates therapist-behaviour classification on the AnnoMI counselling-dialogue corpus using transcript-grouped splits, comparing elastic-net and RoBERTa models with metrics including macro-F1 and accuracy.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences","Finance & Economics"],"primaryDomain":"Health & Life Sciences","industrySectors":["Software & Cloud","Pharma & Biotech","Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-08-21","firstSeenAt":"2026-08-22","recognitionConfidence":0.4,"links":{"report":"https://github.com/abdullahuseyinli-dot/annomi-counselling-dialogue-analysis","pdf":null,"project":null,"code":"https://github.com/abdullahuseyinli-dot/annomi-counselling-dialogue-analysis","data":null,"hfPaper":null},"evidence":{"snippet":"annomi-counselling-dialogue-analysis Transcript-grouped NLP benchmark for therapist-behaviour classification on AnnoMI counselling machine-learning natural-language-processing reproducibility roberta text-classification # AnnoMI Counselling Dialogue Analysis A reproducible NLP benchmark for therapist-behaviour classification on the [AnnoMI](https://github.com/uccollab/AnnoMI) counselling-dialogue corpus.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:abdullahuseyinli-dot/annomi-counselling-dialogue-analysis"},"ranking":{"30d":{"score":23,"rank":135,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":339,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates therapist-behaviour classification on the AnnoMI counselling-dialogue corpus using transcript-grouped splits, comparing elastic-net and RoBERTa models with metrics including macro-F1 and accuracy.","whyItMatters":"Provides reproducible evaluation for counselling dialogue analysis, addressing data leakage in transcript-level splits and offering uncertainty estimates for model comparisons.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"7e9f522352b75099911fa9be8651e32eaec28ef1b6aadd312c0d0610b282701e"},"motivation":"annomi-counselling-dialogue-analysis Transcript-grouped NLP benchmark for therapist-behaviour classification on AnnoMI counselling machine-learning natural-language-processing reproducibility roberta text-classification # AnnoMI Counselling Dialogue Analysis A reproducible NLP benchmark for therapist-behaviour classification on the [AnnoMI](https://github.com/uccollab/AnnoMI) counselling-dialogue corpus.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/abdullahuseyinli-dot/annomi-counselling-dialogue-analysis","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-22T19:19:35.289219Z"},"attentionForecast":{"score":45,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets a niche but active NLP domain with transparent methods and reproducibility, likely drawing moderate attention from researchers in clinical NLP."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"cross-domain"},{"id":"catalog_da4ae39cfdd40aa5","familyId":"catalog_family_da4ae39cfdd40aa5","name":"Anthropic OSS-Fuzz crash","oneLine":"Share of evaluated OSS-Fuzz entry points where the model produced at least a crash.","description":"Share of evaluated OSS-Fuzz entry points where the model produced at least a crash.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.anthropic.com/news/claude-fable-5-mythos-5","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_da4ae39cfdd40aa5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/anthropicossfuzzanycrash"}],"catalogSources":[{"catalog":"benchlm","sourceId":"anthropicOssFuzzAnyCrash","url":"https://benchlm.ai/benchmarks/anthropicossfuzzanycrash","paperUrl":"https://www.anthropic.com/news/claude-fable-5-mythos-5","year":"2026","fullName":"Anthropic OSS-Fuzz Any-Crash Rate","format":"Any-crash rate","tasks":"Approximately 830 OSS-Fuzz entry points","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_265e48430f6e7af4","familyId":"catalog_family_265e48430f6e7af4","name":"Anthropic OSS-Fuzz write primitive","oneLine":"Share of evaluated OSS-Fuzz entry points where the model achieved a write primitive or stronger result.","description":"Share of evaluated OSS-Fuzz entry points where the model achieved a write primitive or stronger result.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.anthropic.com/news/claude-fable-5-mythos-5","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_265e48430f6e7af4"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/anthropicossfuzzwriteprimitive"}],"catalogSources":[{"catalog":"benchlm","sourceId":"anthropicOssFuzzWritePrimitive","url":"https://benchlm.ai/benchmarks/anthropicossfuzzwriteprimitive","paperUrl":"https://www.anthropic.com/news/claude-fable-5-mythos-5","year":"2026","fullName":"Anthropic OSS-Fuzz Write-Primitive-or-Higher Rate","format":"Write-primitive-or-higher rate","tasks":"Approximately 830 OSS-Fuzz entry points","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_antiskillbench_e40657d8","familyId":"bmf_cdeea9a0b393","name":"AntiSkillBench","oneLine":"AntiSkillBench is a benchmark for evaluating privacy leakage and impersonation risks in persona-skills. It includes 7,500 persona-grounded dialogue traces and evaluates across three skill-distillation strategies.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03700","pdf":"https://arxiv.org/pdf/2608.03700","project":"https://yonglixiang.github.io/AntiSkillBench","code":"https://github.com/yonglixiang/AntiSkillBench","data":null,"hfPaper":"https://huggingface.co/papers/2608.03700"},"evidence":{"snippet":"To systematically investigate the safety of the persona-skill pipeline, we introduce AntiSkillBench, an end-to-end benchmark for evaluating risks and defenses across the persona-skill pipeline.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":9,"hfDailySubmittedAt":"2026-08-05T00:00:00.000Z","githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03700"},"ranking":{"30d":{"score":25,"rank":101,"coverage":0.85,"confidence":"High"},"90d":{"score":26,"rank":291,"coverage":0.7,"confidence":"Medium"}},"description":"AntiSkillBench is a benchmark for evaluating privacy leakage and impersonation risks in persona-skills. It includes 7,500 persona-grounded dialogue traces and evaluates across three skill-distillation strategies.","whyItMatters":"Provides an end-to-end evaluation of risks and defenses in the persona-skill pipeline, enabling comparison of model safety across different settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b964d761566aee8f471a2d53c607fb5abc799cb8ebc4c0b781ff6c7c4c9d99f9"},"motivation":"Persona skills distill personal interaction histories into portable and executable artifacts for downstream agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03700","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Yongli Xiang","organizationType":"academic-lab","sourceUrl":"https://yonglixiang.github.io/AntiSkillBench","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_anygroundbench_08e0374e","familyId":"bmf_8d8cdd5ff428","name":"AnyGroundBench","oneLine":"AnyGroundBench evaluates video grounding in vision-language models across five specialized domains (animal, industry, sports, surgery, public security) with spatio-temporal annotations. It provides training and test splits per domain for zero-shot and in-context learning evaluation.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.02269","pdf":"https://arxiv.org/pdf/2607.02269","project":null,"code":"https://github.com/rinost081/AnyGroundBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.02269"},"evidence":{"snippet":"To bridge this gap, we introduce AnyGroundBench, a domain-adaptation benchmark designed to shift the STVG evaluation paradigm from static zero-shot testing to rigorous domain adaptation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":9,"hfDailySubmittedAt":"2026-07-03T00:00:00.000Z","githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02269"},"ranking":{"90d":{"score":33,"rank":211,"coverage":0.7,"confidence":"Medium"}},"description":"AnyGroundBench evaluates video grounding in vision-language models across five specialized domains (animal, industry, sports, surgery, public security) with spatio-temporal annotations. It provides training and test splits per domain for zero-shot and in-context learning evaluation.","whyItMatters":"Addresses the gap between existing benchmark evaluations on general daily-life videos and real-world specialized applications. Provides a structured domain-adaptation protocol to assess model adaptability in specialized fields, enabling comparative evaluation of VLMs in practical scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f4722e2e2d9b2f384f477de3286a945be6cdf4e3c0a41f2234a454164b93a0b0"},"motivation":"Vision-Language Models (VLMs) have demonstrated immense promise in Spatio-Temporal Video Grounding (STVG).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02269","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Keio University","organizationType":"academic-lab","sourceUrl":"https://github.com/rinost081/AnyGroundBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_aor-bench_86ee63c7","familyId":"bmf_b04fbb60d247","name":"AOR-Bench","oneLine":"AOR-Bench is a benchmark of 3,000 pseudo-harmful audio samples across six categories, designed to evaluate over-refusal in large audio language models. It assesses whether models incorrectly reject benign audio queries.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SD"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-19","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.21147","pdf":"https://arxiv.org/pdf/2606.21147","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.21147"},"evidence":{"snippet":"To study this problem, we introduce \\textbf{AOR-Bench} (\\textbf{A}udio \\textbf{O}ver-\\textbf{R}efusal \\textbf{Bench}mark), the first benchmark for over-refusal specifically designed for LALMs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.21147"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AOR-Bench is a benchmark of 3,000 pseudo-harmful audio samples across six categories, designed to evaluate over-refusal in large audio language models. It assesses whether models incorrectly reject benign audio queries.","whyItMatters":"Over-refusal in audio models is understudied. AOR-Bench provides a standardized set to measure this behavior, helping align safety mechanisms with real-world audio context.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a5a17e203a80a3d4a5cef9a08d3c84445573f948d1506a6a3925bb338ca90aff"},"motivation":"Large Audio Language Models (LALMs) have demonstrated strong performance across a wide range of audio tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"EMNLP 2026","evidence":"To appear in EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2606.21147","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"EMNLP 2026","reviewStatus":"accepted","decisionRaw":"To appear in EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.21147","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"To appear in EMNLP 2026","level":"author-claim"}]}],"publishers":[{"name":"AOR-Bench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.21147","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_846be5fc541c5039","familyId":"catalog_family_846be5fc541c5039","name":"Apex","oneLine":"Apex is a challenging frontier reasoning benchmark testing advanced multi-step problem solving across difficult STEM and logical tasks.","description":"Apex is a challenging frontier reasoning benchmark testing advanced multi-step problem solving across difficult STEM and logical tasks.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_846be5fc541c5039"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/apex"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/apex"}],"catalogSources":[{"catalog":"benchlm","sourceId":"apex","url":"https://benchlm.ai/benchmarks/apex","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"Apex","format":"Pass@1 math benchmark","tasks":"Advanced mathematical reasoning","successorKey":null},{"catalog":"llm-stats","sourceId":"apex","url":"https://llm-stats.com/benchmarks/apex","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_b52bb09bba7986a0","familyId":"catalog_family_b52bb09bba7986a0","name":"Apex Shortlist","oneLine":"A shortlist subset of the Apex mathematical reasoning benchmark reported in DeepSeek-V4 model evaluations.","description":"A shortlist subset of the Apex mathematical reasoning benchmark reported in DeepSeek-V4 model evaluations.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b52bb09bba7986a0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/apexshortlist"}],"catalogSources":[{"catalog":"benchlm","sourceId":"apexShortlist","url":"https://benchlm.ai/benchmarks/apexshortlist","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"Apex Shortlist","format":"Pass@1 math benchmark","tasks":"Advanced mathematical reasoning","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_0c16e811aeab795e","familyId":"catalog_family_0c16e811aeab795e","name":"APEX-Agents","oneLine":"APEX-Agents is a benchmark evaluating AI agents on long horizon professional tasks that require sustained reasoning, planning, and execution across complex multi-step workflows.","description":"APEX-Agents is a benchmark evaluating AI agents on long horizon professional tasks that require sustained reasoning, planning, and execution across complex multi-step workflows.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0c16e811aeab795e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/apexagents"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/apex-agents"}],"catalogSources":[{"catalog":"benchlm","sourceId":"apexAgents","url":"https://benchlm.ai/benchmarks/apexagents","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"APEX-Agents","format":"Agent task-completion score","tasks":"Professional-services agent tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"apex-agents","url":"https://llm-stats.com/benchmarks/apex-agents","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","agents"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_269af5b11a2387c6","familyId":"catalog_family_269af5b11a2387c6","name":"APEX-Agents-AA","oneLine":"Artificial Analysis' implementation of the APEX-Agents benchmark for long-horizon professional-services agent tasks.","description":"Artificial Analysis' implementation of the APEX-Agents benchmark for long-horizon professional-services agent tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/apex-agents-aa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_269af5b11a2387c6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/apexagentsaa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"apexAgentsAa","url":"https://benchlm.ai/benchmarks/apexagentsaa","paperUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","year":"2026","fullName":"APEX-Agents-AA","format":"Pass@1","tasks":"452 professional-services agent tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_019660c276df8920","familyId":"catalog_family_019660c276df8920","name":"APEX-SWE","oneLine":"APEX-SWE evaluates AI agents on software engineering tasks requiring multi-step coding, debugging, and verification.","description":"APEX-SWE evaluates AI agents on software engineering tasks requiring multi-step coding, debugging, and verification.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/apex-swe","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_019660c276df8920"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/apex-swe"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"apex-swe","url":"https://llm-stats.com/benchmarks/apex-swe","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_dd09bda504101131","familyId":"catalog_family_dd09bda504101131","name":"API-Bank","oneLine":"A comprehensive benchmark for tool-augmented LLMs that evaluates API planning, retrieval, and calling capabilities. Contains 314 tool-use dialogues with 753 API calls across 73 API tools, designed to assess how effectively LLMs can utilize external tools and overcome obstacles in tool leveraging.","description":"A comprehensive benchmark for tool-augmented LLMs that evaluates API planning, retrieval, and calling capabilities. Contains 314 tool-use dialogues with 753 API calls across 73 API tools, designed to assess how effectively LLMs can utilize external tools and overcome obstacles in tool leveraging.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/api-bank","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dd09bda504101131"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/api-bank"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"api-bank","url":"https://llm-stats.com/benchmarks/api-bank","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","tool calling"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_da715f03ba3dbdb6","familyId":"catalog_family_da715f03ba3dbdb6","name":"App-Bench","oneLine":"A six-task full-stack web-app benchmark that measures how much required functionality an AI builder or coding assistant delivers from one prompt without human code edits.","description":"A six-task full-stack web-app benchmark that measures how much required functionality an AI builder or coding assistant delivers from one prompt without human code edits.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://appbench.ai/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_da715f03ba3dbdb6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/appbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"appBench","url":"https://benchlm.ai/benchmarks/appbench","paperUrl":"https://appbench.ai/","year":"2025","fullName":"App-Bench","format":"Best-of-three one-shot feature completion","tasks":"6 full-stack app-building tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_aps-bench_6b2027af","familyId":"bmf_99de9ecb924b","name":"APS-Bench","oneLine":"APS-Bench is a 50-question QA dataset with auditable gold answers for evaluating retrieval-augmented generation over scientific facility operations knowledge.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Research Infrastructure"],"capabilities":[],"topics":["physics.acc-ph"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.24663","pdf":"https://arxiv.org/pdf/2607.24663","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.24663"},"evidence":{"snippet":"We construct APS-Bench, a 50-question, question-answering (QA) dataset with auditable gold answers.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24663"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"APS-Bench is a 50-question QA dataset with auditable gold answers for evaluating retrieval-augmented generation over scientific facility operations knowledge.","whyItMatters":"It addresses the need for evaluating RAG systems on niche institutional knowledge, but its small scale and facility-specific scope limit broader utility.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3ffc8411e7e3445d17ae4e9458ab9a3c8a7f49ffaf4ba3fda450efe677ef3d2d"},"motivation":"Scientific user facilities accumulate decades of operational knowledge that no single search index covers: electronic logbooks, technical documents, internal wikis, operations chat messages, maintenance records, and live control-system data.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24663","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_arac-benchmarking-auto-research-s-alignmen_57539495","familyId":"bmf_4242a96c003f","name":"ARAC-Bench","oneLine":"ARAC-Bench evaluates auto-research systems on alignment and completeness of research trajectories across proposal, experiment, and synthesis stages using rubrics derived from academic cognition skills.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":0.45,"links":{"report":"https://arxiv.org/abs/2608.12788","pdf":"https://arxiv.org/pdf/2608.12788","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We propose Auto-Research's Alignment and Completeness, ARAC-Bench: a Researcher-Mimicking Evaluation framework that shifts the objective from matching final answers to reproducing high-quality human research processes.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12788"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ARAC-Bench evaluates auto-research systems on alignment and completeness of research trajectories across proposal, experiment, and synthesis stages using rubrics derived from academic cognition skills.","whyItMatters":"Provides a standardized framework for assessing autonomous research systems beyond final output, enabling comparison of research process quality.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"55d6a28570650423d62e1cb5723ea7fc915342a7fad484abc1e2d35ba0cf35c0"},"motivation":"The rapid advancement of Auto-Research has surfaced a fundamental evaluation challenge: how can we measure the alignment, logical coherence, and evolutionary completeness of its research trajectory with human research behavior?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally introduced with a distinctive evaluation framework, but no public code or data link is provided in the input; however, the abstract states it provides a reusable protocol and reward signal.","canonicalNameSource":"abstract","canonicalNameEvidence":"We propose Auto-Research's Alignment and Completeness, ARAC-Bench: a Researcher-Mimicking Evaluation framework that shifts the objective from matching final answers to reproducing high-quality human research processes."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12788","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":45,"confidence":"Low","horizon":"7d","reason":"The topic of auto-research evaluation is novel, but the lack of immediately visible public artifacts may moderate early attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_arbigraph_c1f2133e","familyId":"bmf_ef00fea6017a","name":"ArbiGraph","oneLine":"ArbiGraph is a benchmark generator for evaluating tool-assisted language agents' context management via scalable task graphs with exact automatic verification.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20764","pdf":"https://arxiv.org/pdf/2607.20764","project":null,"code":"https://github.com/pavelgolikov/ArbiGraph.git","data":null,"hfPaper":"https://huggingface.co/papers/2607.20764"},"evidence":{"snippet":"We introduce ARBIGRAPH, a benchmark generator for evaluating whether tool-assisted language agents can retain, update, compose, and discard task-relevant context across extended reasoning workflows.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20764"},"ranking":{"90d":{"score":22,"rank":383,"coverage":0.7,"confidence":"Medium"}},"description":"ArbiGraph is a benchmark generator for evaluating tool-assisted language agents' context management via scalable task graphs with exact automatic verification.","whyItMatters":"Context management is critical for long reasoning workflows; this generator allows controlled variation of task complexity, but the public path is incomplete.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5163da202c15cdd5d75077b82ad4f8400b600dbd40b88577ea86a9150fac2b6a"},"motivation":"We introduce ARBIGRAPH, a benchmark generator for evaluating whether tool-assisted language agents can retain, update, compose, and discard task-relevant context across extended reasoning workflows.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20764","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_404fcfb394d23199","familyId":"catalog_family_404fcfb394d23199","name":"Arc","oneLine":"The Abstraction and Reasoning Corpus (ARC) is a benchmark designed to measure human-like general fluid intelligence through grid-based reasoning tasks. It consists of 800 tasks (400 training, 400 evaluation) where each task presents input-output grids that require understanding abstract patterns and transformations. Test-takers must produce exactly correct output grids for all test inputs in a task to solve it, with 3 trials allowed per test input. ARC aims to enable fair comparisons of general intelligence between AI systems and humans using priors designed to be as close as possible to innate human priors.","description":"The Abstraction and Reasoning Corpus (ARC) is a benchmark designed to measure human-like general fluid intelligence through grid-based reasoning tasks. It consists of 800 tasks (400 training, 400 evaluation) where each task presents input-output grids that require understanding abstract patterns and transformations. Test-takers must produce exactly correct output grids for all test inputs in a task to solve it, with 3 trials allowed per test input. ARC aims to enable fair comparisons of general intelligence between AI systems and humans using priors designed to be as close as possible to innate human priors.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/arc","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_404fcfb394d23199"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/arc"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"arc","url":"https://llm-stats.com/benchmarks/arc","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_220634b4ce946da2","familyId":"catalog_family_220634b4ce946da2","name":"ARC-AGI","oneLine":"The Abstraction and Reasoning Corpus for Artificial General Intelligence (ARC-AGI) is a benchmark designed to test general intelligence and abstract reasoning capabilities through visual grid-based transformation tasks. Each task consists of 2-5 demonstration pairs showing input grids transformed into output grids according to underlying rules, with test-takers required to infer these rules and apply them to novel test inputs. The benchmark uses colored grids (up to 30x30) with 10 discrete colors/symbols, designed to measure human-like general fluid intelligence and skill-acquisition efficiency with minimal prior knowledge.","description":"The Abstraction and Reasoning Corpus for Artificial General Intelligence (ARC-AGI) is a benchmark designed to test general intelligence and abstract reasoning capabilities through visual grid-based transformation tasks. Each task consists of 2-5 demonstration pairs showing input grids transformed into output grids according to underlying rules, with test-takers required to infer these rules and apply them to novel test inputs. The benchmark uses colored grids (up to 30x30) with 10 discrete colors/symbols, designed to measure human-like general fluid intelligence and skill-acquisition efficiency with minimal prior knowledge.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Spatial Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/arc-agi","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_220634b4ce946da2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/arc-agi"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"arc-agi","url":"https://llm-stats.com/benchmarks/arc-agi","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","spatial reasoning","vision"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_4dae675f25ab2857","familyId":"catalog_family_4dae675f25ab2857","name":"ARC-AGI v2","oneLine":"ARC-AGI-2 is an upgraded benchmark for measuring abstract reasoning and problem-solving abilities in AI systems through visual grid transformation tasks. It evaluates fluid intelligence via input-output grid pairs (1x1 to 30x30) using colored cells (0-9), requiring models to identify underlying transformation rules from demonstration examples and apply them to test cases. Designed to be easy for humans but challenging for AI, focusing on core cognitive abilities like spatial reasoning, pattern recognition, and compositional generalization.","description":"ARC-AGI-2 is an upgraded benchmark for measuring abstract reasoning and problem-solving abilities in AI systems through visual grid transformation tasks. It evaluates fluid intelligence via input-output grid pairs (1x1 to 30x30) using colored cells (0-9), requiring models to identify underlying transformation rules from demonstration examples and apply them to test cases. Designed to be easy for humans but challenging for AI, focusing on core cognitive abilities like spatial reasoning, pattern recognition, and compositional generalization.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Spatial Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/arc-agi-v2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4dae675f25ab2857"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/arc-agi-v2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"arc-agi-v2","url":"https://llm-stats.com/benchmarks/arc-agi-v2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","spatial reasoning","vision"],"catalogModelCount":17,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_5f5b31e25b5a8867","familyId":"catalog_family_5f5b31e25b5a8867","name":"ARC-AGI-1","oneLine":"ARC Prize fluid-intelligence benchmark using novel visual grid transformations.","description":"ARC Prize fluid-intelligence benchmark using novel visual grid transformations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arcprize.org/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5f5b31e25b5a8867"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/arcagi1"}],"catalogSources":[{"catalog":"benchlm","sourceId":"arcAgi1","url":"https://benchlm.ai/benchmarks/arcagi1","paperUrl":"https://arcprize.org/","year":"2026","fullName":"ARC-AGI-1 Semi-Private Evaluation","format":"Verified accuracy","tasks":"Semi-private ARC-AGI-1 evaluation set","successorKey":null}],"catalogCategories":["reasoning"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0138a663d691f494","familyId":"catalog_family_0138a663d691f494","name":"ARC-AGI-2","oneLine":"ARC-AGI-2 is the second-generation Abstraction and Reasoning Corpus benchmark measuring fluid, general reasoning and abstraction.","description":"ARC-AGI-2 is the second-generation Abstraction and Reasoning Corpus benchmark measuring fluid, general reasoning and abstraction.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arcprize.org/arc-agi/2/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0138a663d691f494"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/arc-agi-2"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/arcagi2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"arcAgi2","url":"https://benchlm.ai/benchmarks/arc-agi-2","paperUrl":"https://arcprize.org/arc-agi/2/","year":2025,"fullName":"Abstraction and Reasoning Corpus for AGI v2","format":"Grid transformation puzzles with novel rules","tasks":"Visual pattern completion and abstract reasoning","successorKey":null},{"catalog":"llm-stats","sourceId":"arcagi2","url":"https://llm-stats.com/benchmarks/arcagi2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_7d3ff108e044ad47","familyId":"catalog_family_7d3ff108e044ad47","name":"ARC-AGI-3","oneLine":"ARC-AGI-3 is the third-generation Abstraction and Reasoning Corpus benchmark, an interactive-reasoning evaluation designed to measure fluid, novel problem-solving ability that remains far from saturated for frontier models.","description":"ARC-AGI-3 is the third-generation Abstraction and Reasoning Corpus benchmark, an interactive-reasoning evaluation designed to measure fluid, novel problem-solving ability that remains far from saturated for frontier models.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7d3ff108e044ad47"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/arcagi3"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/arc-agi-3"}],"catalogSources":[{"catalog":"benchlm","sourceId":"arcAgi3","url":"https://benchlm.ai/benchmarks/arcagi3","paperUrl":"https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf","year":2026,"fullName":"Abstraction and Reasoning Corpus for AGI v3","format":"Agentic task completion under a capped evaluation budget","tasks":"Interactive game-like tasks with hidden rules","successorKey":null},{"catalog":"llm-stats","sourceId":"arc-agi-3","url":"https://llm-stats.com/benchmarks/arc-agi-3","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_da72f57a8db8bba7","familyId":"catalog_family_da72f57a8db8bba7","name":"ARC-C","oneLine":"The AI2 Reasoning Challenge (ARC) Challenge Set is a multiple-choice question-answering benchmark containing grade-school level science questions that require advanced reasoning capabilities. ARC-C specifically contains questions that were answered incorrectly by both retrieval-based and word co-occurrence algorithms, making it a particularly challenging subset designed to test commonsense reasoning abilities in AI systems.","description":"The AI2 Reasoning Challenge (ARC) Challenge Set is a multiple-choice question-answering benchmark containing grade-school level science questions that require advanced reasoning capabilities. ARC-C specifically contains questions that were answered incorrectly by both retrieval-based and word co-occurrence algorithms, making it a particularly challenging subset designed to test commonsense reasoning abilities in AI systems.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/arc-c","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_da72f57a8db8bba7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/arc-c"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"arc-c","url":"https://llm-stats.com/benchmarks/arc-c","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":34,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0e86e28ad3f67105","familyId":"catalog_family_0e86e28ad3f67105","name":"ARC-E","oneLine":"ARC-E (AI2 Reasoning Challenge - Easy Set) is a subset of grade-school level, multiple-choice science questions that requires knowledge and reasoning capabilities. Part of the AI2 Reasoning Challenge dataset containing 5,197 questions that test scientific reasoning and factual knowledge. The Easy Set contains questions that are answerable by retrieval-based and word co-occurrence algorithms, making them more accessible than the Challenge Set.","description":"ARC-E (AI2 Reasoning Challenge - Easy Set) is a subset of grade-school level, multiple-choice science questions that requires knowledge and reasoning capabilities. Part of the AI2 Reasoning Challenge dataset containing 5,197 questions that test scientific reasoning and factual knowledge. The Easy Set contains questions that are answerable by retrieval-based and word co-occurrence algorithms, making them more accessible than the Challenge Set.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/arc-e","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0e86e28ad3f67105"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/arc-e"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"arc-e","url":"https://llm-stats.com/benchmarks/arc-e","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_archegraph_38554c04","familyId":"bmf_9e9576facb0b","name":"ArchEGraph","oneLine":"ArchEGraph evaluates geometry-topology-physics aligned building energy modeling through graph reconstruction and topology-informed load prediction tasks, with standardized protocols and generalization experiments.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.06772","pdf":"https://arxiv.org/pdf/2608.06772","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.06772"},"evidence":{"snippet":"We present ArchEGraph, a large-scale benchmark dataset that represents buildings as heterogeneous graphs with aligned geometry, topology, weather, and zone-level thermal loads.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06772"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"ArchEGraph evaluates geometry-topology-physics aligned building energy modeling through graph reconstruction and topology-informed load prediction tasks, with standardized protocols and generalization experiments.","whyItMatters":"Provides a large-scale dataset and benchmark tasks for building energy modeling, enabling development and evaluation of surrogate models that couple geometry, topology, and physics.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6d821a6450bba6ab7d634139e126804552f6ab3c729aa12c2b085ba13218dc04"},"motivation":"Accurate estimation of building energy use is essential for achieving carbon neutral and sustainable buildings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06772","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_82bde15e43356e7f","familyId":"catalog_family_82bde15e43356e7f","name":"Arena Hard","oneLine":"Arena-Hard-Auto is an automatic evaluation benchmark for instruction-tuned LLMs consisting of 500 challenging real-world prompts curated by BenchBuilder. It includes open-ended software engineering problems, mathematical questions, and creative writing tasks. The benchmark uses LLM-as-a-Judge methodology with GPT-4.1 and Gemini-2.5 as automatic judges to approximate human preference. Arena-Hard achieves 98.6% correlation with human preference rankings and provides 3x higher separation of model performances compared to MT-Bench, making it highly effective for distinguishing between models of similar quality.","description":"Arena-Hard-Auto is an automatic evaluation benchmark for instruction-tuned LLMs consisting of 500 challenging real-world prompts curated by BenchBuilder. It includes open-ended software engineering problems, mathematical questions, and creative writing tasks. The benchmark uses LLM-as-a-Judge methodology with GPT-4.1 and Gemini-2.5 as automatic judges to approximate human preference. Arena-Hard achieves 98.6% correlation with human preference rankings and provides 3x higher separation of model performances compared to MT-Bench, making it highly effective for distinguishing between models of similar quality.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General","Creativity","Writing"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/arena-hard","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_82bde15e43356e7f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/arena-hard"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"arena-hard","url":"https://llm-stats.com/benchmarks/arena-hard","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","creativity","writing"],"catalogModelCount":26,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0f07233db8031494","familyId":"catalog_family_0f07233db8031494","name":"Arena-Hard v2","oneLine":"Arena-Hard-Auto v2 is a challenging benchmark consisting of 500 carefully curated prompts sourced from Chatbot Arena and WildChat-1M, designed to evaluate large language models on real-world user queries. The benchmark covers diverse domains including open-ended software engineering problems, mathematics, creative writing, and technical problem-solving. It uses LLM-as-a-Judge for automatic evaluation, achieving 98.6% correlation with human preference rankings while providing 3x higher separation of model performances compared to MT-Bench. The benchmark emphasizes prompt specificity, complexity, and domain knowledge to better distinguish between model capabilities.","description":"Arena-Hard-Auto v2 is a challenging benchmark consisting of 500 carefully curated prompts sourced from Chatbot Arena and WildChat-1M, designed to evaluate large language models on real-world user queries. The benchmark covers diverse domains including open-ended software engineering problems, mathematics, creative writing, and technical problem-solving. It uses LLM-as-a-Judge for automatic evaluation, achieving 98.6% correlation with human preference rankings while providing 3x higher separation of model performances compared to MT-Bench. The benchmark emphasizes prompt specificity, complexity, and domain knowledge to better distinguish between model capabilities.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General","Creativity","Writing"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/arena-hard-v2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0f07233db8031494"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/arena-hard-v2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"arena-hard-v2","url":"https://llm-stats.com/benchmarks/arena-hard-v2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","creativity","writing"],"catalogModelCount":16,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_7e1829fbc34027db","familyId":"catalog_family_7e1829fbc34027db","name":"ARKitScenes","oneLine":"ARKitScenes evaluates 3D scene understanding and spatial reasoning in AR/VR contexts.","description":"ARKitScenes evaluates 3D scene understanding and spatial reasoning in AR/VR contexts.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Spatial Reasoning","3D","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/arkitscenes","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7e1829fbc34027db"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/arkitscenes"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"arkitscenes","url":"https://llm-stats.com/benchmarks/arkitscenes","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["spatial reasoning","3d","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_armnetbench_46ca6487","familyId":"bmf_4ef71bcd8184","name":"ArmnetBench","oneLine":"A benchmark for robot manipulation policies evaluated on a fleet of low-cost SO-101 cells. It compares 7 policies across 12 tasks in single-arm and bimanual configurations, with 2,518 policy rollouts and 600 reference demonstrations, all labeled successful, suboptimal, or failure. Data is released in LeRobot and RoboMeter formats.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.24481","pdf":"https://arxiv.org/pdf/2607.24481","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.24481"},"evidence":{"snippet":"We introduce ArmnetBench v0.1, a benchmark run on a fleet of low-cost SO-101 cells under light on-site supervision.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24481"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark for robot manipulation policies evaluated on a fleet of low-cost SO-101 cells. It compares 7 policies across 12 tasks in single-arm and bimanual configurations, with 2,518 policy rollouts and 600 reference demonstrations, all labeled successful, suboptimal, or failure. Data is released in LeRobot and RoboMeter formats.","whyItMatters":"Real-world evaluation of manipulation policies is costly and difficult to standardize. This benchmark provides a shared, public protocol with quality-labeled data, enabling comparable assessment of policies and supporting research on learning from mixed-quality demonstrations.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a5f574ec5d10c999c1cc8bac092f80a050edc49005a54385e59a5a2e52da5215"},"motivation":"Real-world evaluation is a bottleneck in developing generalist robot manipulation policies.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24481","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ArmnetBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.24481","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_arteculture_b1d1e3cf","familyId":"bmf_ced67ccbede0","name":"ArtECulture","oneLine":"ArtECulture is a benchmark for culture-conditioned visual emotion understanding with 6,792 artworks and culture-specific labels across three cultures. It evaluates MLLMs on predicting cultural emotion perception and explaining rationale.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03358","pdf":"https://arxiv.org/pdf/2608.03358","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03358"},"evidence":{"snippet":"Thus, we present ArtECulture, a benchmark containing 6,792 artworks with culture-specific emotion labels and explanations across English, Chinese, and Arabic cultures, with balanced Western and non-Western content.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03358"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ArtECulture is a benchmark for culture-conditioned visual emotion understanding with 6,792 artworks and culture-specific labels across three cultures. It evaluates MLLMs on predicting cultural emotion perception and explaining rationale.","whyItMatters":"Provides a balanced benchmark for evaluating cultural variations in visual emotion understanding, enabling comparison of MLLM capabilities across cultures.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"26e56b24950b033194ffdcc47620298b9868078a1bc1ee3b29c58b972e83a34e"},"motivation":"Existing visual emotion understanding methods typically ignore cultural variations in emotional perception.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03358","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_588d925488858d2e","familyId":"catalog_family_588d925488858d2e","name":"Artifacts Bench","oneLine":"Artifacts Bench evaluates a model's ability to generate visual code artifacts, measuring the quality of generated interactive and visual front-end outputs from natural-language requests.","description":"Artifacts Bench evaluates a model's ability to generate visual code artifacts, measuring the quality of generated interactive and visual front-end outputs from natural-language requests.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Frontend Development","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/artifacts-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_588d925488858d2e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/artifacts-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"artifacts-bench","url":"https://llm-stats.com/benchmarks/artifacts-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["frontend development","code"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_373b3921ce6ac401","familyId":"catalog_family_373b3921ce6ac401","name":"Artificial Analysis","oneLine":"Artificial Analysis benchmark evaluates AI models across quality, speed, and pricing dimensions, providing a composite assessment of model capabilities for real-world usage.","description":"Artificial Analysis benchmark evaluates AI models across quality, speed, and pricing dimensions, providing a composite assessment of model capabilities for real-world usage.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/artificial-analysis","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_373b3921ce6ac401"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/artificial-analysis"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"artificial-analysis","url":"https://llm-stats.com/benchmarks/artificial-analysis","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["general"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0e73adf5343720b6","familyId":"catalog_family_0e73adf5343720b6","name":"Artificial Analysis Intelligence Index","oneLine":"A display-only intelligence index published by Artificial Analysis that aggregates provider-reported and benchmark-derived signals into a single model-level score.","description":"A display-only intelligence index published by Artificial Analysis that aggregates provider-reported and benchmark-derived signals into a single model-level score.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0e73adf5343720b6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/artificialanalysis"}],"catalogSources":[{"catalog":"benchlm","sourceId":"artificialAnalysis","url":"https://benchlm.ai/benchmarks/artificialanalysis","paperUrl":"https://artificialanalysis.ai/","year":"2026","fullName":"Artificial Analysis Intelligence Index","format":"Aggregated model score","tasks":"Cross-benchmark intelligence index","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_3ea1a5e4de61cb92","familyId":"catalog_family_3ea1a5e4de61cb92","name":"ArXivMath","oneLine":"ArXivMath is a final-answer benchmark of research-level mathematics maintained by MathArena. Problems are extracted monthly from recent arXiv paper abstracts, then filtered through automated and manual checks to ensure they are self-contained, non-trivial, and verifiable. Because problems are drawn from active research, the benchmark is more realistic and more closely connected to mathematical research than contest or olympiad benchmarks.","description":"ArXivMath is a final-answer benchmark of research-level mathematics maintained by MathArena. Problems are extracted monthly from recent arXiv paper abstracts, then filtered through automated and manual checks to ensure they are self-contained, non-trivial, and verifiable. Because problems are drawn from active research, the benchmark is more realistic and more closely connected to mathematical research than contest or olympiad benchmarks.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/arxivmath","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3ea1a5e4de61cb92"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/arxivmath"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"arxivmath","url":"https://llm-stats.com/benchmarks/arxivmath","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_80634567dcc125d1","familyId":"catalog_family_80634567dcc125d1","name":"ArXivMath Jun. 2026 (no tools)","oneLine":"Final-answer research mathematics problems drawn from recent arXiv abstracts.","description":"Final-answer research mathematics problems drawn from recent arXiv abstracts.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_80634567dcc125d1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/arxivmathjune2026"}],"catalogSources":[{"catalog":"benchlm","sourceId":"arxivMathJune2026","url":"https://benchlm.ai/benchmarks/arxivmathjune2026","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"ArXivMath June 2026 without tools","format":"Final-answer accuracy","tasks":"49 recent research-mathematics problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_da21fe49de825d45","familyId":"catalog_family_da21fe49de825d45","name":"ArXivMath Jun. 2026 (tools)","oneLine":"Final-answer research mathematics problems drawn from recent arXiv abstracts with tool access.","description":"Final-answer research mathematics problems drawn from recent arXiv abstracts with tool access.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_da21fe49de825d45"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/arxivmathjune2026withtools"}],"catalogSources":[{"catalog":"benchlm","sourceId":"arxivMathJune2026WithTools","url":"https://benchlm.ai/benchmarks/arxivmathjune2026withtools","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"ArXivMath June 2026 with tools","format":"Final-answer accuracy with tools","tasks":"49 recent research-mathematics problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_asi-bench_faf0dd90","familyId":"bmf_3b83443d21ef","name":"ASI-Bench","oneLine":"ASI-Bench evaluates AI systems on 60 project-level scientific research tasks across 11 domains, with four guidance levels B1-B4 measuring autonomous execution and innovation.","area":"Agents & Tool Use","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Research & Development"],"capabilities":["Scientific discovery","Research planning","Autonomous execution"],"topics":["AI Scientist","Agents","Reasoning"],"construction":"Hybrid","annotation":"Expert Generated","readiness":"Runnable","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.17271","pdf":"https://arxiv.org/pdf/2608.17271","project":"https://asibench.apexin.ai","code":"https://github.com/apexin-ai/ASI-Bench","data":"https://huggingface.co/datasets/Apexintelligence-AI/ASI-Bench-seed31415","hfPaper":"https://huggingface.co/papers/2608.17271"},"evidence":{"snippet":"ASI-Bench contains 60 project-level research tasks across 11 scientific domains and progressively reduces methodological guidance to test whether AI can independently select methods, conduct research, and produce verifiable results.","reasonCodes":["named benchmark release","public evaluator","submission path","leaderboard"]},"dataStatus":"primary-source-reviewed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":62,"hfDailySubmittedAt":"2026-08-19T00:00:00.000Z","githubStars":275,"githubScope":"benchmark_repo","hfDatasetDownloads":3197,"hfDatasetLikes":19},"source":{"type":"arxiv","id":"2608.17271"},"ranking":{"30d":{"score":81,"rank":1,"coverage":1.0,"confidence":"High","datasetDownloadRank":3,"datasetRankPopulation":30},"90d":{"score":73,"rank":5,"coverage":1.0,"confidence":"High","datasetDownloadRank":8,"datasetRankPopulation":66}},"description":"ASI-Bench evaluates AI systems on 60 project-level scientific research tasks across 11 domains, with four guidance levels B1-B4 measuring autonomous execution and innovation.","whyItMatters":"It addresses the gap in evaluating AI's independent scientific exploration and execution, revealing current dependence on human guidance for end-to-end research.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4ced4b176bd1572a50abbafd269d41d66f3dce2c7bc8ccc509a202924efd679e"},"motivation":"Evaluate whether AI agents can independently select methods, execute end-to-end research, and produce verifiable scientific results as human methodological guidance is progressively withdrawn.","constructionDetail":"The maintainers distilled more than 1,300 candidate research ideas through five review rounds, over 1,100 review assignments, more than 2,000 task revisions, and over 1,500 sandbox runs.","metrics":[{"name":"Tasks","value":"60","note":"project-level tasks"},{"name":"Coverage","value":"11","note":"scientific domains"},{"name":"Protocol","value":"B1–B4","note":"progressively less methodological guidance"},{"name":"Evaluation","value":"Expert + execution","note":"cross-review, AI-assisted audit, sandbox runs, scorer validation"}],"detail":{"taskBreakdown":["Mathematics","Physics","Chemistry","Biology","Astronomy","Materials science","Earth science","Medicine & biostatistics","Computer science","Robotics","Electrical engineering"],"protocol":{"tasks":60,"conditions":["B1","B2","B3","B4"],"primaryMetric":"Scientific Score","comparisonMetric":"B3 Scientific Score","aggregation":"Macro-average over tasks","runs":"Three independent runs unless marked otherwise","tools":"Paper results reported without external tool access"},"modelCoverage":[{"provider":"OpenAI","models":["GPT-5.5","GPT-5.6 Sol"]},{"provider":"Anthropic","models":["Claude Opus 4.8","Claude Opus 5"]},{"provider":"Moonshot AI","models":["Kimi K2.7","Kimi K3"]},{"provider":"Z.AI","models":["GLM-5.2","GLM-5.3"]},{"provider":"DeepSeek","models":["DeepSeek V4 Flash","DeepSeek V4 Pro"]},{"provider":"MiniMax","models":["MiniMax M3"]},{"provider":"Xiaomi","models":["MiMo V2.5 Pro"]}],"modelCoverageNote":"These systems were evaluated by the benchmark authors; their presence does not by itself mean the model provider adopted or endorsed ASI-Bench.","adoption":{"independentOrganizations":[],"note":"No source-linked independent model-provider adoption has been recorded yet."},"leaderboard":{"primaryMetric":"B3 Scientific Score","bestScore":51.6,"bestSystem":"GPT-5.6 Sol (ultra) · Codex","meanScore":26.62,"evaluatedConfigurations":18,"saturationStatus":"Not saturated","assessment":"Benchmark-author assessment","asOf":"2026-08-18","sourceUrl":"https://arxiv.org/abs/2608.17271"},"leaderboardUrl":"https://asibench.apexin.ai/leaderboard","submissionUrl":"https://asibench.apexin.ai/submit"},"curation":{"state":"source-reviewed","reviewedAt":"2026-08-19","sources":["https://github.com/apexin-ai/ASI-Bench","https://arxiv.org/abs/2608.17271"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17271","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-20T00:00:00Z"},"releaseDates":{"firstPublicAt":"2026-08-18","paperV1At":"2026-08-18"},"evaluationMode":"score_submission","availability":{"paperStatus":"available","githubStatus":"available","hfDatasetStatus":"available","evaluatorStatus":"available","submissionStatus":"available"},"publishers":[{"name":"Apex Intelligence AI","organizationType":"company-research-lab","sourceUrl":"https://github.com/apexin-ai/ASI-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_asob-bench_b6e17903","familyId":"bmf_5360c6d29e43","name":"ASOB-Bench","oneLine":"ASOB-Bench evaluates diffusion classifiers along three bias dimensions: attribute binding, size-order bias, and background dependency, using newly constructed datasets and extending existing frameworks.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.03831","pdf":"https://arxiv.org/pdf/2607.03831","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.03831"},"evidence":{"snippet":"We introduce ASOB-Bench, a bias evaluation for diffusion classifiers along three dimensions: Attribute binding, Size-Order bias, and Background dependency.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.03831"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ASOB-Bench evaluates diffusion classifiers along three bias dimensions: attribute binding, size-order bias, and background dependency, using newly constructed datasets and extending existing frameworks.","whyItMatters":"This probe reveals distinct bias profiles in diffusion classifiers compared to vision-language models, informing robustness improvements in diffusion-based systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ef93e04d95b5c0a37ea91b183df23af0236be39ceb60d8c47fe800405fe2705f"},"motivation":"Diffusion models have recently been repurposed for zero-shot classification, giving rise to diffusion classifiers that identify the best-matching text prompt by minimizing the noise-prediction error.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.03831","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_asr-model-benchmark_50d41822","familyId":"bmf_8795f3a05ca1","name":"ASR Model Benchmark","oneLine":"Compares Whisper Small, Faster-Whisper Small, and Wav2Vec2 Base on 20 clean and 20 noisy LibriSpeech samples using WER, inference latency, RTF, and memory usage on CPU.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/Anuragkokate09/asr-model-benchmark","pdf":null,"project":null,"code":"https://github.com/Anuragkokate09/asr-model-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"asr-model-benchmark Reproducible CPU benchmark comparing Whisper Small, Faster-Whisper Small, and Wav2Vec2 under clean and controlled noisy speech conditions.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:anuragkokate09/asr-model-benchmark"},"ranking":{"30d":{"score":23,"rank":128,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":332,"coverage":0.55,"confidence":"Low"}},"description":"Compares Whisper Small, Faster-Whisper Small, and Wav2Vec2 Base on 20 clean and 20 noisy LibriSpeech samples using WER, inference latency, RTF, and memory usage on CPU.","whyItMatters":"Provides a reproducible CPU-based comparison of popular ASR models under controlled noise, offering practical guidance for resource-constrained deployment.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"81864529753333728acaeedfb9928a3f6272f47feb843b22663fc94a482c5ff8"},"motivation":"asr-model-benchmark Reproducible CPU benchmark comparing Whisper Small, Faster-Whisper Small, and Wav2Vec2 under clean and controlled noisy speech conditions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/Anuragkokate09/asr-model-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":38,"confidence":"Medium","horizon":"7d","reason":"ASR model comparison is a common interest, but the small dataset and CPU-only setup limit its novelty and expected early attention."},"evaluationMode":"score_submission","publishers":[{"name":"Anurag Kokate","organizationType":"community","sourceUrl":"https://github.com/Anuragkokate09/asr-model-benchmark","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_assertllm2_9bd7c97e","familyId":"bmf_b6287b8c059f","name":"AssertLLM2","oneLine":"Evaluates LLM generation of SystemVerilog assertions from structured design specifications and RTL, across two tasks: bug-prevention and bug-hunting. The benchmark includes 83 real-world designs with golden and mutated RTL, and assesses syntactic validity, formal provability, coverage, and mutation-based bug detection.","area":"Language & Knowledge","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Semiconductors"],"capabilities":[],"topics":["cs.AR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27472","pdf":"https://arxiv.org/pdf/2605.27472","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27472"},"evidence":{"snippet":"To address these limitations, we introduce AssertLLM2, an open-source benchmark for realistic assertion generation in hardware verification.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27472"},"ranking":{},"description":"Evaluates LLM generation of SystemVerilog assertions from structured design specifications and RTL, across two tasks: bug-prevention and bug-hunting. The benchmark includes 83 real-world designs with golden and mutated RTL, and assesses syntactic validity, formal provability, coverage, and mutation-based bug detection.","whyItMatters":"Fills a gap in realistic evaluation for assertion generation by using full specifications and buggy RTL, supporting comparison of LLM capabilities for hardware verification tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1fab1e04aac1f930b13d9a1b88636933d24d18d71520fe38d92588727e084c71"},"motivation":"Assertion-based verification (ABV) is a cornerstone of modern hardware design, yet manually translating design intent into formal SystemVerilog Assertions (SVAs) remains labor-intensive and error-prone.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27472","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_astromind_587e5d1f","familyId":"bmf_f3195093bff0","name":"AstroMind","oneLine":"AstroMind evaluates LLM reasoning about spacecraft behavior across intent inference, maneuver parameter estimation, and threat assessment, using physics-grounded simulations with realistic sensing noise. Metrics capture semantic correctness and quantitative consistency under physical constraints.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.24573","pdf":"https://arxiv.org/pdf/2605.24573","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24573"},"evidence":{"snippet":"AstroMind is a physics-grounded benchmark designed to close that gap.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24573"},"ranking":{},"description":"AstroMind evaluates LLM reasoning about spacecraft behavior across intent inference, maneuver parameter estimation, and threat assessment, using physics-grounded simulations with realistic sensing noise. Metrics capture semantic correctness and quantitative consistency under physical constraints.","whyItMatters":"Space situational awareness lacks benchmarks for reasoning about why spacecraft maneuver, not just detection. AstroMind provides a shared test that combines physics accuracy and tactical interpretation, enabling comparison of models on this critical reasoning task.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f248c443fec22a71ed76885eb2b752d4b2739d32e164f4b1dbde09b60c6c2669"},"motivation":"Understanding why a spacecraft maneuvers -- rather than simply that it did -- is an increasingly important problem for space domain awareness as Earth orbits grow crowded and contested.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24573","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_asuka-bench_f32387bf","familyId":"bmf_6edc4f2a50c6","name":"Asuka-Bench","oneLine":"Evaluates code agents on 50 web tasks with underspecified intent and multi-round refinement via browser-rendered behavior.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05920","pdf":"https://arxiv.org/pdf/2606.05920","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05920"},"evidence":{"snippet":"We present Asuka-Bench, a benchmark that pairs underspecified user intent with multi-round refinement, grounded in browser-rendered behavior.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05920"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates code agents on 50 web tasks with underspecified intent and multi-round refinement via browser-rendered behavior.","whyItMatters":"Could address gaps in code-generation benchmarks by testing iterative refinement, but lacks public artifacts for reuse.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d097e873c47845fb787d0946ea792aeca785aaa9cde536bde1b04ccc02ea6451"},"motivation":"Existing code-generation benchmarks score a single mapping from a complete prompt to a one-shot output.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05920","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_asynctool_407c214b","familyId":"bmf_ec6315a36e48","name":"AsyncTool","oneLine":"AsyncTool is a benchmark for evaluating asynchronous function calling in multi-task tool-use environments. It presents multiple tasks with simulated tool response latency, assessing step-level tool-call correctness, sub-task completion, and task-level end-to-end success, along with efficiency-oriented metrics.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27995","pdf":"https://arxiv.org/pdf/2605.27995","project":null,"code":"https://github.com/StoKou/repo-asynctool","data":null,"hfPaper":"https://huggingface.co/papers/2605.27995"},"evidence":{"snippet":"To evaluate it, we propose AsyncTool, a benchmark for assessing LLM-based agents in interactive multi-task tool-use environments with delayed tool feedback.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":16,"hfDailySubmittedAt":null,"githubStars":101,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27995"},"ranking":{},"description":"AsyncTool is a benchmark for evaluating asynchronous function calling in multi-task tool-use environments. It presents multiple tasks with simulated tool response latency, assessing step-level tool-call correctness, sub-task completion, and task-level end-to-end success, along with efficiency-oriented metrics.","whyItMatters":"Real-world tool use involves delays, but existing benchmarks assume immediate responses. AsyncTool measures whether agents can coordinate multiple tasks and use idle time efficiently, identifying key failure modes for temporal reasoning and task coordination.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"95896a723774ae6cbdf64037db9c11076b9462fd539755d0e5c13089ecf1e6f0"},"motivation":"Large language model (LLM)-based agents have shown strong capabilities in using external tools to solve complex tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27995","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AsyncTool Team","organizationType":"academic-lab","sourceUrl":"https://github.com/StoKou/repo-asynctool","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_ateliereval_cc1f5b0c","familyId":"bmf_40ece635281f","name":"AtelierEval","oneLine":"AtelierEval evaluates prompting proficiency of humans and MLLMs for text-to-image systems via 360 tasks, using a skill-based agentic evaluator (AtelierJudge) that scores prompt-image pairs subjectively and objectively.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22645","pdf":"https://arxiv.org/pdf/2605.22645","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.22645"},"evidence":{"snippet":"We introduce AtelierEval, the first unified benchmark that quantifies prompting proficiency across 360 expert-crafted tasks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.22645"},"ranking":{},"description":"AtelierEval evaluates prompting proficiency of humans and MLLMs for text-to-image systems via 360 tasks, using a skill-based agentic evaluator (AtelierJudge) that scores prompt-image pairs subjectively and objectively.","whyItMatters":"Introduces a new evaluation angle for T2I pipelines, measuring upstream prompting ability which is currently unassessed.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"849add50d73fcb34de0a5434ed011db8056f58f81af2ee4ea6b4c4d5062dea81"},"motivation":"Text-to-image (T2I) systems increasingly rely on upstream prompters, either humans or multimodal large language models (MLLMs), to translate user intent into detailed prompts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026","evidence":"Accepted by ICML 2026","evidenceUrl":"https://arxiv.org/abs/2605.22645","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026","reviewStatus":"accepted","decisionRaw":"Accepted by ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2605.22645","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by ICML 2026","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_7c82602500857aa6","familyId":"catalog_family_7c82602500857aa6","name":"atlas","oneLine":"Catalog-listed benchmark; original-source verification is pending.","description":"Catalog-listed benchmark; original-source verification is pending.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":[],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/community:ed90e889-4678-4fbd-98ab-0e654f4bf35e","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7c82602500857aa6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/community:ed90e889-4678-4fbd-98ab-0e654f4bf35e"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"community:ed90e889-4678-4fbd-98ab-0e654f4bf35e","url":"https://llm-stats.com/benchmarks/community:ed90e889-4678-4fbd-98ab-0e654f4bf35e","datasetSlug":"atlas","versionCount":0,"subsetCount":0,"rowCount":null,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":[],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_atmoscoder-bench_24ddfa7c","familyId":"bmf_5582793e0564","name":"AtmosCoder-Bench","oneLine":"AtmosCoder-Bench is an execution-grounded benchmark for LLMs on atmospheric science computation, with 436 problems and 3,910 variants, grading by executing code solutions against ground truth.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-19","firstSeenAt":"2026-08-20","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.18726","pdf":"https://arxiv.org/pdf/2608.18726","project":null,"code":"https://github.com/acodercat/AtmosCoder-Bench","data":null,"hfPaper":null},"evidence":{"snippet":"Here we introduce AtmosCoder-Bench, an execution-grounded benchmark that makes the calculation process visible.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.18726"},"ranking":{"30d":{"score":31,"rank":74,"coverage":0.55,"confidence":"Low"},"90d":{"score":31,"rank":229,"coverage":0.55,"confidence":"Low"}},"description":"AtmosCoder-Bench is an execution-grounded benchmark for LLMs on atmospheric science computation, with 436 problems and 3,910 variants, grading by executing code solutions against ground truth.","whyItMatters":"Existing evaluations score final answers, overlooking calculation process. AtmosCoder-Bench makes process visible, revealing failures in applying formulas and adapting to regimes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"024f39fc515d74250d85c364f676811509d1beaf126fe6dc0035b30611fd08e9"},"motivation":"Large language models are increasingly used for quantitative work in the environmental sciences, yet existing evaluations score only final answers, leaving calculation process unobserved.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-20T09:45:58.504563Z","model":"deepseek-v4-flash","decisionReason":"The benchmark defines a repeatable evaluation object with a clear scoring contract (execution-grounded, unit-aware grading), publicly available code and result data, and a credible path for other teams to run it."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.18726","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-20T09:44:47.308583Z"},"evaluationMode":"score_submission","publishers":[{"name":"acodercat","organizationType":"community","sourceUrl":"https://github.com/acodercat/AtmosCoder-Bench","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_atom-bench_82c6474b","familyId":"bmf_728bb9ed779a","name":"ATOM-Bench","oneLine":"ATOM-Bench evaluates atomic skills and compositional generalization in manipulation policies across 30 atomic tasks and 24 held-out compositional tasks, using paired single-arm and dual-arm robot tracks, with 3,000 human demonstrations and evaluation rollout data released.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.16826","pdf":"https://arxiv.org/pdf/2606.16826","project":"https://flageval-baai.github.io/AtomBenchPage","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.16826"},"evidence":{"snippet":"We introduce \\textbf{ATOM-Bench}, a real-world benchmark for evaluating both atomic skills and compositional generalization in manipulation policies.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16826"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ATOM-Bench evaluates atomic skills and compositional generalization in manipulation policies across 30 atomic tasks and 24 held-out compositional tasks, using paired single-arm and dual-arm robot tracks, with 3,000 human demonstrations and evaluation rollout data released.","whyItMatters":"ATOM-Bench provides a public diagnostic testbed for disentangling failures in motor execution, instruction grounding, and compositional reuse, which is crucial for advancing generalist manipulation policies beyond demonstrated tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a7d424463565a060cc14b7d35ed2016941916957d47ea8ff6f0f77684b046304"},"motivation":"Generalist manipulation policies are increasingly presented as foundation models for robotic control, but their real-world generalization remains difficult to diagnose.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.16826","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"BAAI","organizationType":"company-research-lab","sourceUrl":"https://flageval-baai.github.io/AtomBenchPage","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_32cc95bd5668cb3f","familyId":"catalog_family_32cc95bd5668cb3f","name":"Atomic evasion","oneLine":"Irregular's domain-level evaluation of cybersecurity evasion capability.","description":"Irregular's domain-level evaluation of cybersecurity evasion capability.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.irregular.com/research/assessing-gpt-5.6-sol","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_32cc95bd5668cb3f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/atomicevasion"}],"catalogSources":[{"catalog":"benchlm","sourceId":"atomicEvasion","url":"https://benchlm.ai/benchmarks/atomicevasion","paperUrl":"https://www.irregular.com/research/assessing-gpt-5.6-sol","year":"2026","fullName":"Atomic Evasion","format":"Domain average","tasks":"Atomic cyber tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d3b7d8b1692e7fa0","familyId":"catalog_family_d3b7d8b1692e7fa0","name":"Atomic network attacks","oneLine":"Irregular's domain-level evaluation of network attack simulation capability.","description":"Irregular's domain-level evaluation of network attack simulation capability.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.irregular.com/research/assessing-gpt-5.6-sol","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d3b7d8b1692e7fa0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/atomicnetworkattacksimulation"}],"catalogSources":[{"catalog":"benchlm","sourceId":"atomicNetworkAttackSimulation","url":"https://benchlm.ai/benchmarks/atomicnetworkattacksimulation","paperUrl":"https://www.irregular.com/research/assessing-gpt-5.6-sol","year":"2026","fullName":"Atomic Network Attack Simulation","format":"Domain average","tasks":"Atomic cyber tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_a1da8a393209195a","familyId":"catalog_family_a1da8a393209195a","name":"Atomic vulnerability research","oneLine":"Irregular's domain-level evaluation of vulnerability research and exploitation capability.","description":"Irregular's domain-level evaluation of vulnerability research and exploitation capability.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.irregular.com/research/assessing-gpt-5.6-sol","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a1da8a393209195a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/atomicvulnerabilityresearch"}],"catalogSources":[{"catalog":"benchlm","sourceId":"atomicVulnerabilityResearch","url":"https://benchlm.ai/benchmarks/atomicvulnerabilityresearch","paperUrl":"https://www.irregular.com/research/assessing-gpt-5.6-sol","year":"2026","fullName":"Atomic Vulnerability Research and Exploitation","format":"Domain average","tasks":"Atomic cyber tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_atrex-bench_ab05aff2","familyId":"bmf_068d0e6d2ab8","name":"Atrex-Bench","oneLine":"Evaluates coding agents on generating GPU kernels from PyTorch references across 30 operators and 440 shapes derived from production inference traces. Scoring uses a three-stage evaluator measuring compile success, numerical correctness, and speed-of-light (SOL) efficiency against a cached roofline. The benchmark includes 4 DSL backends and supports multi-vendor GPUs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Agents"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.14541","pdf":"https://arxiv.org/pdf/2607.14541","project":null,"code":"https://github.com/alibaba/atrex-bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.14541"},"evidence":{"snippet":"We present Atrex-Bench, a benchmark whose 30 operators and 440 shapes are sampled directly from full-cluster production inference traces of compute-limited, memory-rich GPUs.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":28,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14541"},"ranking":{"90d":{"score":40,"rank":149,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates coding agents on generating GPU kernels from PyTorch references across 30 operators and 440 shapes derived from production inference traces. Scoring uses a three-stage evaluator measuring compile success, numerical correctness, and speed-of-light (SOL) efficiency against a cached roofline. The benchmark includes 4 DSL backends and supports multi-vendor GPUs.","whyItMatters":"Prior GPU kernel benchmarks draw from synthetic or curated sources that diverge from deployed workloads. This benchmark provides a production-trace-driven evaluation that emphasizes operators consuming the most serving time, enabling assessment of agent performance on relevant tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"63e87ad19eac0773073fb3036c263af57bac1f75c8689e70d1acf14ca0e070ce"},"motivation":"Existing GPU kernel generation benchmarks draw problems from synthetic or curated sources that diverge from deployed workloads.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14541","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Alibaba","organizationType":"company-research-lab","sourceUrl":"https://github.com/alibaba/atrex-bench","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_669c5f1c9d6e93a2","familyId":"catalog_family_669c5f1c9d6e93a2","name":"AttaQ","oneLine":"AttaQ is a unique dataset containing adversarial examples in the form of questions designed to provoke harmful or inappropriate responses from large language models. The benchmark evaluates safety vulnerabilities by using specialized clustering techniques that analyze both the semantic similarity of input attacks and the harmfulness of model responses, facilitating targeted improvements to model safety mechanisms.","description":"AttaQ is a unique dataset containing adversarial examples in the form of questions designed to provoke harmful or inappropriate responses from large language models. The benchmark evaluates safety vulnerabilities by using specialized clustering techniques that analyze both the semantic similarity of input attacks and the harmfulness of model responses, facilitating targeted improvements to model safety mechanisms.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/attaq","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_669c5f1c9d6e93a2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/attaq"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"attaq","url":"https://llm-stats.com/benchmarks/attaq","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_auau_d78e1117","familyId":"bmf_bcfe7d2ed921","name":"AuAu","oneLine":"AuAu combines psychometric instruments, vignettes, and realistic prompts to assess authoritarian tendencies in LLM responses, measuring sub-concepts like aggression, submission, and conventionalism.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.16127","pdf":"https://arxiv.org/pdf/2606.16127","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.16127"},"evidence":{"snippet":"We introduce AuAu, a comprehensive benchmark for assessing the risk of authoritarian tendencies in LLM responses.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16127"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"AuAu combines psychometric instruments, vignettes, and realistic prompts to assess authoritarian tendencies in LLM responses, measuring sub-concepts like aggression, submission, and conventionalism.","whyItMatters":"Provides a structured approach to auditing LLM authoritarian alignment, highlighting variations across models and the impact of system prompts on authoritarian output.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e73080a9d4422eb95d8fc245d6576a9d02d6f57e509aecebb04d3cf0516373f7"},"motivation":"The worldwide rise of authoritarianism and the growing role of Large Language Models (LLMs) in users' everyday lives raise the question of whether specific models exhibit or promote authoritarian attitudes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.16127","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_authmem-bench_56d03df5","familyId":"bmf_3b53359afe7a","name":"AuthMem-Bench","oneLine":"AuthMem-Bench evaluates authority collapse in persistent memory for LLM agents. It uses a paired benchmark holding claims and tasks fixed while varying source authority, measuring write-time collapse, authorization errors, and automatic authority preservation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Self-Evolution","Agents","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.01679","pdf":"https://arxiv.org/pdf/2608.01679","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01679"},"evidence":{"snippet":"We introduce AuthMem-Bench, a controlled paired benchmark that holds the focal claim and downstream task fixed while varying only source authority.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01679"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AuthMem-Bench evaluates authority collapse in persistent memory for LLM agents. It uses a paired benchmark holding claims and tasks fixed while varying source authority, measuring write-time collapse, authorization errors, and automatic authority preservation.","whyItMatters":"Memory consolidation can erase authority constraints, leading to unauthorized actions. AuthMem-Bench provides a controlled benchmark to measure and improve authority preservation in memory systems, relevant for safe agent deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"59f3b5e566de1dbc5839708d614c163fef369422470391756f0e748c095c89a6"},"motivation":"Persistent memory allows (self-evolving) LLM agents to adapt across tasks by consolidating heterogeneous interaction histories into reusable facts, preferences, observations, and rules.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.01679","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_authoritybench_5d25d571","familyId":"bmf_fba66cb2685f","name":"AuthorityBench","oneLine":"AuthorityBench evaluates how citation-based authority signals affect epistemic behavior in LLMs using 220,564 prompts across four domains, with a 2x2 factorial design crossing claim veracity and citation veracity.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13104","pdf":"https://arxiv.org/pdf/2606.13104","project":null,"code":"https://github.com/floating-reeds/AuthorityBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.13104"},"evidence":{"snippet":"We introduce AuthorityBench, a 220,564-prompt multi-domain benchmark that isolates how citation-based authority signals influence epistemic behavior in LLMs.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13104"},"ranking":{"90d":{"score":23,"rank":381,"coverage":0.55,"confidence":"Low"}},"description":"AuthorityBench evaluates how citation-based authority signals affect epistemic behavior in LLMs using 220,564 prompts across four domains, with a 2x2 factorial design crossing claim veracity and citation veracity.","whyItMatters":"The benchmark isolates citation presence from content to measure susceptibility to citation-induced hallucination, providing a standardized protocol for assessing reliability in citation-augmented settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cc87a45f373a49769ef8548228e8198ada140e0c2bb0e4c064a8cf4fd98198d3"},"motivation":"Large language models are increasingly deployed in citation-augmented settings, yet the effect of citation presence on model behavior independent of factual content remains poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"AI4GOOD and EIML at ICML 2026","evidence":"10 pages, 5 figures. Accepted to AI4GOOD and EIML at ICML 2026","evidenceUrl":"https://arxiv.org/abs/2606.13104","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"AI4GOOD and EIML at ICML 2026","reviewStatus":"accepted","decisionRaw":"10 pages, 5 figures. Accepted to AI4GOOD and EIML at ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.13104","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"10 pages, 5 figures. Accepted to AI4GOOD and EIML at ICML 2026","level":"author-claim"}]}],"publishers":[{"name":"floating-reeds","organizationType":"community","sourceUrl":"https://github.com/floating-reeds/AuthorityBench","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_authtrace_cea4d208","familyId":"bmf_fe5b1fd5a9d7","name":"AuthTrace","oneLine":"AuthTrace is a diagnostic benchmark for evidence construction in thematically dense single-author corpora, with fan-in annotations and pack-level protocol.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.25382","pdf":"https://arxiv.org/pdf/2605.25382","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.25382"},"evidence":{"snippet":"We introduce AuthTrace, a diagnostic benchmark built on thematically dense single-author corpora where near-miss distractors share style, topic, and vocabulary with the required evidence.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25382"},"ranking":{},"description":"AuthTrace is a diagnostic benchmark for evidence construction in thematically dense single-author corpora, with fan-in annotations and pack-level protocol.","whyItMatters":"It provides a diagnostic lens for identifying where evidence construction fails and which paradigm works best.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a947c9fbcad978fba5a4e9fd32cd721cf97546ed3efb60e4488e21e0a145fa54"},"motivation":"Evidence construction--the stage that determines which passages reach the language model before generation begins--is evaluated paradigm by paradigm, leaving practitioners with no principled way to diagnose which organization strategy fails, where, or why.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25382","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_842c223a65e7195d","familyId":"catalog_family_842c223a65e7195d","name":"AutoCAD-Bench","oneLine":"A Markov Studios computer-use benchmark that asks agents to produce 2D drawings and 3D models in AutoCAD.","description":"A Markov Studios computer-use benchmark that asks agents to produce 2D drawings and 3D models in AutoCAD.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.markovstudios.com/research/autocad-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_842c223a65e7195d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/autocadbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"autoCadBench","url":"https://benchlm.ai/benchmarks/autocadbench","paperUrl":"https://www.markovstudios.com/research/autocad-bench","year":"2026","fullName":"AutoCAD-Bench","format":"Task completion rate at a 75-point rubric threshold","tasks":"21 2D drawing tasks and 29 3D modeling tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_autolab_db01499e","familyId":"bmf_5d290b897878","name":"AutoLab","oneLine":"AutoLab evaluates frontier models on long-horizon closed-loop optimization tasks across system optimization, CUDA kernel optimization, model development, and puzzle challenges, with 36 expert-curated tasks and a strict wall-clock budget.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05080","pdf":"https://arxiv.org/pdf/2606.05080","project":"https://autolab.moe/","code":"https://github.com/autolabhq/autolab","data":null,"hfPaper":"https://huggingface.co/papers/2606.05080"},"evidence":{"snippet":"To address this gap, we introduce AutoLab, a new benchmark for ultra long-horizon closed-loop optimization.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":31,"hfDailySubmittedAt":null,"githubStars":164,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05080"},"ranking":{"90d":{"score":63,"rank":16,"coverage":0.7,"confidence":"Medium"}},"description":"AutoLab evaluates frontier models on long-horizon closed-loop optimization tasks across system optimization, CUDA kernel optimization, model development, and puzzle challenges, with 36 expert-curated tasks and a strict wall-clock budget.","whyItMatters":"Fills the gap in evaluating sustained iterative improvement in agents, moving beyond single-turn or short-horizon benchmarks to test persistence and empirical feedback incorporation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"29758b55619444afc386cde48a9130b280acb0afc841bff305d51e0ece3b7059"},"motivation":"Scientific and engineering progress is fundamentally a long-horizon iterative process: proposing changes, running experiments, measuring outcomes, and continuously refining artifacts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05080","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_83aeecd3f494da49","familyId":"catalog_family_83aeecd3f494da49","name":"AutoLogi","oneLine":"AutoLogi is an automated method for synthesizing open-ended logic puzzles to evaluate reasoning abilities of Large Language Models. The benchmark addresses limitations of existing multiple-choice reasoning evaluations by featuring program-based verification and controllable difficulty levels. It includes 1,575 English and 883 Chinese puzzles, enabling more reliable evaluation that better distinguishes models' reasoning capabilities across languages.","description":"AutoLogi is an automated method for synthesizing open-ended logic puzzles to evaluate reasoning abilities of Large Language Models. The benchmark addresses limitations of existing multiple-choice reasoning evaluations by featuring program-based verification and controllable difficulty levels. It includes 1,575 English and 883 Chinese puzzles, enabling more reliable evaluation that better distinguishes models' reasoning capabilities across languages.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/autologi","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_83aeecd3f494da49"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/autologi"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"autologi","url":"https://llm-stats.com/benchmarks/autologi","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e59105b35af353d8","familyId":"catalog_family_e59105b35af353d8","name":"AutomationBench","oneLine":"AutomationBench is a tool-use benchmark that evaluates AI agents on automating real-world workflows, testing their ability to orchestrate tools and complete multi-step automation tasks.","description":"AutomationBench is a tool-use benchmark that evaluates AI agents on automating real-world workflows, testing their ability to orchestrate tools and complete multi-step automation tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agentic","Reasoning","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e59105b35af353d8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/automationbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/automationbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"automationBench","url":"https://benchlm.ai/benchmarks/automationbench","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"AutomationBench","format":"Agent task-completion score","tasks":"600 public automation tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"automationbench","url":"https://llm-stats.com/benchmarks/automationbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","agents","tool calling"],"catalogModelCount":14,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_530d0ae739bef10b","familyId":"catalog_family_530d0ae739bef10b","name":"AutomationBench-AA","oneLine":"AutomationBench-AA is Artificial Analysis's independently run version of AutomationBench, covering 657 real-world SaaS workflow tasks across 40 simulated applications (e.g. Gmail, Slack, Salesforce, HubSpot). It scores the share of objectives an agent completes without violating business guardrails, using a private held-out task set.","description":"AutomationBench-AA is Artificial Analysis's independently run version of AutomationBench, covering 657 real-world SaaS workflow tasks across 40 simulated applications (e.g. Gmail, Slack, Salesforce, HubSpot). It scores the share of objectives an agent completes without violating business guardrails, using a private held-out task set.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/automationbench-aa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_530d0ae739bef10b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/automationbench-aa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"automationbench-aa","url":"https://llm-stats.com/benchmarks/automationbench-aa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","tool calling"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_automedbench_61e5be52","familyId":"bmf_bbd322eeddb0","name":"AutoMedBench","oneLine":"AutoMedBench evaluates autonomous AI agents on end-to-end medical-AI research tasks spanning segmentation, image enhancement, VQA, report generation, and lesion detection. Tasks follow a five-stage workflow (Plan, Setup, Validate, Inference, Submit), with scoring based on both final task performance and stage-level rubric scores.","area":"Agents & Tool Use","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Agents","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01961","pdf":"https://arxiv.org/pdf/2606.01961","project":null,"code":"https://github.com/AutoMedBench/AutoMedBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.01961"},"evidence":{"snippet":"To address this gap, we present AutoMedBench, a workflow-aware benchmark for autonomous medical-AI research across diverse medical imaging and multimodal inference tasks, organizing agent execution into a unified five-stage workflow (S1-S5): Plan, Setup, Validate, Inference, and Submit.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":28,"hfDailySubmittedAt":null,"githubStars":58,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01961"},"ranking":{},"description":"AutoMedBench evaluates autonomous AI agents on end-to-end medical-AI research tasks spanning segmentation, image enhancement, VQA, report generation, and lesion detection. Tasks follow a five-stage workflow (Plan, Setup, Validate, Inference, Submit), with scoring based on both final task performance and stage-level rubric scores.","whyItMatters":"Existing medical agent benchmarks focus on final outputs, obscuring failure points. AutoMedBench provides granular stage-level scoring and error analysis, enabling targeted assessment of agent capabilities and highlighting bottlenecks like verification and submission, which can guide improvement priorities in automated medical research systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8b76663e82520ad458808d2fdc47e2e21c699407edea64cc4bc7e6c2b7db4452"},"motivation":"Autonomous agents are increasingly expected to support end-to-end medical-AI research workflows, moving beyond isolated prediction tasks or short-form clinical question answering.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01961","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_autoworldmodel-bench_01d33f6e","familyId":"bmf_c9aa949f276e","name":"AutoWorldModel-Bench","oneLine":"AutoWorldModel-Bench evaluates coding agents on autonomously improving a base world model under fixed compute budget across eight game environments. Uses structured-state representation, with held-out test split and closed-loop iteration.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.11216","pdf":"https://arxiv.org/pdf/2608.11216","project":"https://electronicarts.github.io/AutoWorldModelBench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.11216"},"evidence":{"snippet":"We introduce AutoWorldModel-Bench, a closed-loop benchmark in which frontier coding agents autonomously improve a provided base world model under a fixed compute budget.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":13,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11216"},"ranking":{"90d":{"score":50,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"AutoWorldModel-Bench evaluates coding agents on autonomously improving a base world model under fixed compute budget across eight game environments. Uses structured-state representation, with held-out test split and closed-loop iteration.","whyItMatters":"Addresses the gap in agent benchmarks for open-ended research tasks, enabling comparison of agents on research-like workflows rather than engineering-to-spec tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5494aae2e8fbd6ee1e24e3e85e999963339b379d2b11d751c975eac82f277ae8"},"motivation":"World modeling is an unsettled field: architectures, training objectives, and state representations interact in complex ways, and no single recipe dominates across environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11216","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Electronic Arts","organizationType":"company-research-lab","sourceUrl":"https://electronicarts.github.io/AutoWorldModelBench/","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_av-syncbench_20190a72","familyId":"bmf_e3adb616e734","name":"AV-SyncBench","oneLine":"AV-SyncBench evaluates audio-visual synchronization by separating temporal and semantic consistency. It contains 3,269 videos and 38,390 samples across 10 scenarios and 5 tasks, with data filtered and verified for on-screen sound sources.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.00726","pdf":"https://arxiv.org/pdf/2607.00726","project":"https://fgt7t6g.github.io/AV-SyncBench","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.00726"},"evidence":{"snippet":"We propose AV-SyncBench, the first benchmark to fully separate temporal and semantic evaluation for audio-visual synchronization.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.00726"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AV-SyncBench evaluates audio-visual synchronization by separating temporal and semantic consistency. It contains 3,269 videos and 38,390 samples across 10 scenarios and 5 tasks, with data filtered and verified for on-screen sound sources.","whyItMatters":"Existing AV feature extraction evaluations are coupled, preventing independent assessment of temporal and semantic alignment. AV-SyncBench provides a decoupled benchmark to quantify feature quality for alignment and downstream tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"34dda38356605d3475f21bd7a31babfcb03e38ef2bf133e7362bcc77ae5f0f2a"},"motivation":"Audio-visual feature extraction is a fundamental component of multimodal understanding and generation tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"Interspeech 2026","evidence":"Accepted by Interspeech 2026","evidenceUrl":"https://arxiv.org/abs/2607.00726","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"Interspeech 2026","reviewStatus":"accepted","decisionRaw":"Accepted by Interspeech 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.00726","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by Interspeech 2026","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_avalanchebench_c5aa4b06","familyId":"bmf_970926f050ea","name":"AvalancheBench","oneLine":"AvalancheBench evaluates enterprise data agents on latent world recovery, scoring analytical understanding of segments, drivers, temporal events, and relationships from generated observations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.DB"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.24183","pdf":"https://arxiv.org/pdf/2605.24183","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24183"},"evidence":{"snippet":"We introduce AvalancheBench, a benchmark for evaluating enterprise data agents through \\emph{latent world recovery}.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24183"},"ranking":{},"description":"AvalancheBench evaluates enterprise data agents on latent world recovery, scoring analytical understanding of segments, drivers, temporal events, and relationships from generated observations.","whyItMatters":"The evaluation gap is the need for controlled diagnostics of whether agents recover analytical structure behind enterprise data, with a rubric for partial credit and error propagation analysis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5bf9af3041eb34538b42c92e5c23e6bd35f14f5971bc4b3454c667256e0db942"},"motivation":"We introduce AvalancheBench, a benchmark for evaluating enterprise data agents through \\emph{latent world recovery}.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24183","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_avalon-tom-bench_a3c88250","familyId":"bmf_370b127bbad7","name":"Avalon-ToM-Bench","oneLine":"Avalon-ToM-Bench evaluates fine-grained theory of mind in LLMs using a 2x2 taxonomy of epistemic/motivational reasoning crossed with inference/action, via human-crafted perspective-constrained queries from The Resistance: Avalon.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.09638","pdf":"https://arxiv.org/pdf/2608.09638","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09638"},"evidence":{"snippet":"We present Avalon-ToM-Bench, a fine-grained benchmark that operationalizes ToM through the asymmetric-information mechanics of The Resistance: Avalon.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09638"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Avalon-ToM-Bench evaluates fine-grained theory of mind in LLMs using a 2x2 taxonomy of epistemic/motivational reasoning crossed with inference/action, via human-crafted perspective-constrained queries from The Resistance: Avalon.","whyItMatters":"It provides a diagnostic decomposition of ToM abilities, distinguishing reasoning, expression, and policy, and offers insights into training and inference interventions for improving social reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e7e97550c1421ec4e82d853d9f652f9020d620a4fa0cf9319826d7e8a2b35e6c"},"motivation":"Theory of Mind (ToM) is essential for agent interactions, yet existing evaluations either rely on static scenarios that oversimplify mental-state reasoning or interactive settings that provide limited diagnostic insight.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09638","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_avbench_ce8c8830","familyId":"bmf_edaffec7807d","name":"AVBench","oneLine":"AVBench evaluates audio-video generative models across ten fine-grained dimensions covering visual quality, audio quality, and cross-modal consistency for human-centric scenarios. It uses specialized evaluators trained via preference learning and provides continuous scores from prediction confidence.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.24652","pdf":"https://arxiv.org/pdf/2605.24652","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24652"},"evidence":{"snippet":"To address these issues, we introduce AVBench, a fully automated benchmark tailored for human-centric AV generation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24652"},"ranking":{},"description":"AVBench evaluates audio-video generative models across ten fine-grained dimensions covering visual quality, audio quality, and cross-modal consistency for human-centric scenarios. It uses specialized evaluators trained via preference learning and provides continuous scores from prediction confidence.","whyItMatters":"Existing benchmarks for AV generation are coarse and rely on generic multimodal LLMs, leading to inaccurate assessments. AVBench offers automated, human-aligned evaluation with fine-grained metrics, enabling reliable model comparison and serving as a potential reward signal for RLHF.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c2a081dc2cfddf81dcb28c0d41671c9feedc190ff6f47141701fe425db4215e7"},"motivation":"Rapid advances in audio-video (AV) generation have enabled high-fidelity synthesis with synchronized sound, particularly for human-related scenarios involving speech and interactions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24652","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_ave-compass_477a3826","familyId":"bmf_7977347d6b84","name":"AVE-Compass","oneLine":"AVE-Compass evaluates audio-visual editing models on 145 source videos and 196 instructions with 2,688 checklist items, scoring Instruction Following, Fidelity Preserving, Realism, and Editing Intent via MLLM judging and automated metrics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.MM"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.24821","pdf":"https://arxiv.org/pdf/2607.24821","project":null,"code":"https://github.com/NJU-LINK/AVE-Compass","data":null,"hfPaper":"https://huggingface.co/papers/2607.24821"},"evidence":{"snippet":"We introduce AVE-Compass, a comprehensive benchmark with 145 curated source videos, 196 audio-visually coupled editing instructions, and 2,688 fine-grained checklist items.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":18,"hfDailySubmittedAt":null,"githubStars":9,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24821"},"ranking":{"90d":{"score":40,"rank":150,"coverage":0.7,"confidence":"Medium"}},"description":"AVE-Compass evaluates audio-visual editing models on 145 source videos and 196 instructions with 2,688 checklist items, scoring Instruction Following, Fidelity Preserving, Realism, and Editing Intent via MLLM judging and automated metrics.","whyItMatters":"This benchmark addresses the gap in evaluating coordinated audio-visual edits, providing a structured way to measure cross-modal consistency and non-target preservation, which is critical for advancing real-world video editing systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"40d4b0b484cd6f1adade0e713f9f134250f79ecdab3a5792daef44fac5d87f71"},"motivation":"While instruction-based video editing has advanced rapidly, real-world videos contain tightly coupled audio and visual signals, and editing one modality often requires coordinated changes in the other.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24821","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"NJU-LINK","organizationType":"academic-lab","sourceUrl":"https://github.com/NJU-LINK/AVE-Compass","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_avi-bench_16159c13","familyId":"bmf_b71e730eed92","name":"AVI-Bench","oneLine":"AVI-Bench evaluates omni-multimodal LLMs on audio-visual tasks across perception, understanding, and reasoning stages, with an extension probing primitive audio-visual sensation using unfamiliar low-semantic stimuli.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07643","pdf":"https://arxiv.org/pdf/2606.07643","project":"https://fudancvl.github.io/AVI-Bench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07643"},"evidence":{"snippet":"We introduce AVI-Bench, a cognitively inspired benchmark that evaluates Omni-MLLMs across three stages, perception, understanding, and reasoning, through cross-modal tasks requiring joint audio-visual interpretation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07643"},"ranking":{},"description":"AVI-Bench evaluates omni-multimodal LLMs on audio-visual tasks across perception, understanding, and reasoning stages, with an extension probing primitive audio-visual sensation using unfamiliar low-semantic stimuli.","whyItMatters":"Provides a structured way to measure joint audio-visual capabilities in multimodal models, which is currently under-evaluated.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"01ad0a58264e4b4cff6b1e56d3583d66a3ad3c244f024bc0848fa314a3887ea3"},"motivation":"Recent advances in Omni-Multimodal Large Language Models (Omni-MLLMs) have enabled strong integration of vision, audio, and language.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07643","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_when-writing-style-drifts-benchmarking-aut_3da16bfb","familyId":"bmf_4640bbb5ce1a","name":"AVShift","oneLine":"AVShift evaluates authorship verification across over 150,000 German text pairs spanning three genres and 21 years, with controlled cross-genre, temporal, and AI-era distribution shifts.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":0.45,"links":{"report":"https://arxiv.org/abs/2608.17979","pdf":"https://arxiv.org/pdf/2608.17979","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce AVShift, the first German benchmark for systematically evaluating AV under multiple distribution shifts.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17979"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"AVShift evaluates authorship verification across over 150,000 German text pairs spanning three genres and 21 years, with controlled cross-genre, temporal, and AI-era distribution shifts.","whyItMatters":"Existing authorship verification benchmarks study distribution shifts in isolation and focus on English, so AVShift provides a unified German resource to assess robustness under realistic writing-style drifts.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"c370c0fdfef85b26b0bc459d3d3cb4ce2a1d7793decb3b8e880d8113cba00e21"},"motivation":"Authorship verification (AV) assumes that an author's writing style remains sufficiently stable to distinguish it from that of other writers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"AVShift is formally named in the abstract as a benchmark with a released dataset and code, defining a stable evaluation protocol for authorship verification under distribution shifts.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce AVShift, the first German benchmark for systematically evaluating AV under multiple distribution shifts."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17979","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":42,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses multiple distribution shifts in a non-English language, likely drawing moderate initial interest from the authorship verification community.","reasoning":"Focus on German, explicit shifts, and code release support moderate attention; topic is moderately broad."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_axis_d184e18d","familyId":"bmf_d39c118729fe","name":"AXIS","oneLine":"AXIS is a community-driven data engine and benchmark for robot manipulation, providing browser-based teleoperation for data collection, automated task generation and validation, and a dataset of 207 tasks and 50K+ trajectories, with a systematic held-out protocol for policy evaluation.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.21588","pdf":"https://arxiv.org/pdf/2607.21588","project":"https://axisaiorg.github.io/AXIS-V1/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.21588"},"evidence":{"snippet":"We present AXIS, a growable community-driven data engine and benchmark for scalable robot learning, which enables browser-based teleoperation for large-scale demonstration collection, automatically generates and validates new manipulation tasks, and transforms community-collected demonstrations into training-ready data through automated success checking, quality filtering, trajectory smoothing, and visual and physics-based augmentation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.21588"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"AXIS is a community-driven data engine and benchmark for robot manipulation, providing browser-based teleoperation for data collection, automated task generation and validation, and a dataset of 207 tasks and 50K+ trajectories, with a systematic held-out protocol for policy evaluation.","whyItMatters":"Scaling robot learning requires diverse data and standardized evaluation; AXIS offers a growable pipeline and unified evaluation suite, enabling comparison of VLA policies and studying scaling behavior.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d0ee8d9fcfc697abd47c25111f1da6802acfd57eb6b45f1e34e1013bf926d10d"},"motivation":"Learning effective robot manipulation policies requires diverse, high-quality demonstrations, yet existing data pipelines are often difficult to scale because they rely on specialized hardware, centralized operators, or fixed task suites.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.21588","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AXIS Team","organizationType":"academic-lab","sourceUrl":"https://axisaiorg.github.io/AXIS-V1/","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_babeljudge_6d1b19e2","familyId":"bmf_625004021f92","name":"BabelJudge","oneLine":"BabelJudge audits LLM-as-a-judge reliability across languages and agent trajectories, measuring position bias, verbosity bias, order inconsistency, and cross-lingual degradation without human labels, providing a composite bias-penalised reliability score.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22329","pdf":"https://arxiv.org/pdf/2606.22329","project":null,"code":"https://github.com/Shreyaskc/BabelJudge","data":null,"hfPaper":"https://huggingface.co/papers/2606.22329"},"evidence":{"snippet":"We introduce BabelJudge, an open-source benchmark and reliability audit framework that measures all four failure modes -- position bias, verbosity bias, order inconsistency, and cross-lingual degradation -- on any judge model, without requiring human preference labels.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22329"},"ranking":{"90d":{"score":28,"rank":279,"coverage":0.55,"confidence":"Low"}},"description":"BabelJudge audits LLM-as-a-judge reliability across languages and agent trajectories, measuring position bias, verbosity bias, order inconsistency, and cross-lingual degradation without human labels, providing a composite bias-penalised reliability score.","whyItMatters":"It addresses the gap that raw accuracy hides systematic judge biases, offering a standardized method to quantify reliability for automated evaluation, crucial for trustworthy model comparisons and training data quality in multilingual and agentic contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"54e9b4bc732f37a5829e2fff89667a81889f14664841bb73afe9a6bd155c5849"},"motivation":"LLM-as-a-judge has become the dominant approach to scalable evaluation in NLP pipelines, yet judges themselves carry systematic biases that raw accuracy hides: they favor responses placed in slot A (position bias), they prefer longer responses regardless of quality (verbosity bias), and their reliability degrades sharply in lower-resource languages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22329","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_f7192a6a6674f631","familyId":"catalog_family_f7192a6a6674f631","name":"BabyVision","oneLine":"A benchmark for early-stage visual reasoning and perception on child-like vision tasks.","description":"A benchmark for early-stage visual reasoning and perception on child-like vision tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f7192a6a6674f631"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/babyvision"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/babyvision"}],"catalogSources":[{"catalog":"benchlm","sourceId":"babyVision","url":"https://benchlm.ai/benchmarks/babyvision","paperUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","year":"2026","fullName":"BabyVision","format":"Multimodal visual reasoning","tasks":"Visual perception tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"babyvision","url":"https://llm-stats.com/benchmarks/babyvision","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","vision"],"catalogModelCount":11,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_664ae8314b540447","familyId":"catalog_family_664ae8314b540447","name":"BabyVision w/ Python","oneLine":"A Python-assisted BabyVision evaluation for fine-grained visual perception and grounded reasoning.","description":"A Python-assisted BabyVision evaluation for fine-grained visual perception and grounded reasoning.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_664ae8314b540447"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/babyvisionpython"}],"catalogSources":[{"catalog":"benchlm","sourceId":"babyVisionPython","url":"https://benchlm.ai/benchmarks/babyvisionpython","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"BabyVision with Python","format":"Tool-augmented multimodal score","tasks":"Visual perception tasks with Python","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_backendforge_1bb8c499","familyId":"bmf_6ecda39e082a","name":"BackendForge","oneLine":"BackendForge is a benchmark of 56 contract-defined backend generation tasks from real open-source applications. LLMs must generate Dockerized services evaluated through HTTP tests against an OpenAPI contract.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation"],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.11042","pdf":"https://arxiv.org/pdf/2607.11042","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.11042"},"evidence":{"snippet":"We introduce BackendForge, a benchmark of 56 contract-defined backend generation tasks rewritten from real open-source applications.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.11042"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"BackendForge is a benchmark of 56 contract-defined backend generation tasks from real open-source applications. LLMs must generate Dockerized services evaluated through HTTP tests against an OpenAPI contract.","whyItMatters":"Agentic LLMs need to produce deployable and behaviorally correct software artifacts. BackendForge provides a deterministic, black-box evaluation of backend service generation, exposing gaps between local API implementation and complete service delivery.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"740147521fd047ec3109764fbe7e21679177245b00575ec9f9939387fa9b9ada"},"motivation":"Large language models (LLMs) are increasingly used in agentic coding settings, where they can inspect files, execute commands, run tests, observe failures, and iteratively revise code.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.11042","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_balms_57758cdd","familyId":"bmf_17b4729fcca2","name":"BALMS","oneLine":"Mental health assessment relies on episodic self-report scales, which convert subjective states such as stress into numerical scores but provide only sparse snapshots of wellbeing.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.27219","pdf":"https://arxiv.org/pdf/2608.27219","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.27219"},"evidence":{"snippet":"To address this gap, we introduce BALMS, the first systematic benchmark of LLM-based agentic systems for longitudinal mental health sensing.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27219"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"Mental health assessment relies on episodic self-report scales, which convert subjective states such as stress into numerical scores but provide only sparse snapshots of wellbeing.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 Main Conference","evidence":"Accepted to EMNLP 2026 Main Conference","evidenceUrl":"https://arxiv.org/abs/2608.27219","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"EMNLP 2026 Main Conference","reviewStatus":"accepted","decisionRaw":"Accepted to EMNLP 2026 Main Conference","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.27219","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to EMNLP 2026 Main Conference","level":"author-claim"}]}],"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_register-shifts-break-llm-safety-a-bengali_9b6dd4e8","familyId":"bmf_e32f99ce3fcc","name":"BanglaSafe","oneLine":"Evaluates LLM safety on 879 Bengali prompts spanning 17 harm categories and five prompting conditions varying language, writing style, and authority framing.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-25","recognitionConfidence":0.45,"links":{"report":"http://arxiv.org/abs/2608.22335v1","pdf":"https://arxiv.org/pdf/2608.22335v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce BanglaSafe, a benchmark of 879 Bengali prompts combining 309 natively authored prompts with 570 expert-reviewed prompts, spanning 17 culturally grounded harm categories and five prompting conditions that vary language, writing style, and authority framing.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22335"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLM safety on 879 Bengali prompts spanning 17 harm categories and five prompting conditions varying language, writing style, and authority framing.","whyItMatters":"Provides culturally grounded safety evaluation for Bengali, a widely spoken but underrepresented language in LLM safety testing, and reveals sensitivity to writing style.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"b1e237a6b94a51d418979ab59f8aac6faf2372c75b152fece31d7f160d9cd39e"},"motivation":"Bengali is the seventh-most-spoken language globally, yet LLM safety evaluation remains overwhelmingly English-centric.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is introduced in the paper and includes scored evaluations, but no public code, dataset, or project URL is provided, so the public reuse path is unclear.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce BanglaSafe, a benchmark of 879 Bengali prompts"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22335v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":35,"confidence":"Low","horizon":"7d","reason":"The benchmark targets a significant language and reports striking findings, but lack of public artifacts may limit initial engagement."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_banglaveilguard_8771f924","familyId":"bmf_c5b0b6014460","name":"BanglaVeilGuard","oneLine":"Evaluates Bangla LLM safety across six language forms using 2,366 prompts and a held-out 354-prompt split spanning unsafe, safe, and safe-sensitive requests, with deterministic response scoring and guardrail screening.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-22","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.21880v1","pdf":"https://arxiv.org/pdf/2608.21880v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"This paper presents BanglaVeilGuard, a compact Bangla-first safety benchmark and lightweight prompt guard for six language forms: standard Bangla, Romanized Bangla, Banglish, code-mixed Bangla--English, noisy Bangla, and dialectal Bangla.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.21880"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates Bangla LLM safety across six language forms using 2,366 prompts and a held-out 354-prompt split spanning unsafe, safe, and safe-sensitive requests, with deterministic response scoring and guardrail screening.","whyItMatters":"Addresses the lack of Bangla-specific safety evaluation by covering script variation and code-mixing that English-centric benchmarks miss, providing a repeatable protocol for measuring attack success and guardrail recall.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"c25142a5d0304904d46a8009c5fb6e2c14a4c827e9a5f137b9d780c816c68584"},"motivation":"Bangla large language model (LLM) safety is difficult to evaluate with English-centric or standard-script benchmarks because Bangla users routinely write across scripts, spellings, code-mixed forms, and regional registers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The paper declares a named benchmark with a held-out evaluation split, deterministic scoring, and explicit statement of release for corpus and code, meeting the criteria for a reusable benchmark.","canonicalNameSource":"paper_title","canonicalNameEvidence":"BanglaVeilGuard: Cross-Script Safety Benchmarking and Lightweight Guardrails for Bangla Large Language Models"},"publication":{"status":"acceptance_claimed","venue":"4th International Conference on Computing Advancements (ICCA 2026)","evidence":"Accepted at the 4th International Conference on Computing Advancements (ICCA 2026). 8 pages","evidenceUrl":"http://arxiv.org/abs/2608.21880v1","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-25T17:37:25.889905Z"},"venueAttempts":[{"venueName":"4th International Conference on Computing Advancements (ICCA 2026)","reviewStatus":"accepted","decisionRaw":"Accepted at the 4th International Conference on Computing Advancements (ICCA 2026). 8 pages","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"http://arxiv.org/abs/2608.21880v1","observedAt":"2026-08-25T17:37:25.889905Z","rawValue":"Accepted at the 4th International Conference on Computing Advancements (ICCA 2026). 8 pages","level":"author-claim"}]}],"attentionForecast":{"score":48,"confidence":"Low","horizon":"7d","reason":"The niche focus on Bangla LLM safety and absence of artifact links may limit immediate attention, though the cross-script scope and guardrail results could draw modest interest."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_banglawild_bb99b93a","familyId":"bmf_14ca929f6ebd","name":"BanglaWild","oneLine":"BanglaWild is a benchmark of 2,535 Bengali scene text images with verbatim gold transcriptions and diagnostic attributes. It evaluates OCR and vision-language models on in-the-wild scene text recognition.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03884","pdf":"https://arxiv.org/pdf/2608.03884","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03884"},"evidence":{"snippet":"To address this gap, we introduce BANGLAWILD, a benchmark of 2,535 Bengali scene text images, each paired with a verbatim gold transcription, two categorical axes, four diagnostic attributes, and an orthographically standard form where the in-image text deviates from canonical spelling.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03884"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"BanglaWild is a benchmark of 2,535 Bengali scene text images with verbatim gold transcriptions and diagnostic attributes. It evaluates OCR and vision-language models on in-the-wild scene text recognition.","whyItMatters":"Fills the gap of measuring in-the-wild Bengali scene text recognition, providing a common testbed for OCR and VLMs and enabling analysis of error types.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"310fbe67f6a81ed656d1d8417f49aba8997fa40d7fb29735aa05d8a0c8d04162"},"motivation":"In-the-wild Bengali scene text recognition is largely unmeasured: existing resources target handwritten documents or constrained sign-board parsing, report only aggregate edit-distance metrics, and evaluate either conventional OCR or VLMs, never both on the same in-the-wild data.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03884","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_9dcbfe996a77cf11","familyId":"catalog_family_9dcbfe996a77cf11","name":"BankerToolBench","oneLine":"BankerToolBench is a public benchmark that evaluates models on banking and finance tool-use tasks. Models are scored against dataset rubrics, measuring their ability to correctly invoke tools and complete multi-step financial workflows.","description":"BankerToolBench is a public benchmark that evaluates models on banking and finance tool-use tasks. Models are scored against dataset rubrics, measuring their ability to correctly invoke tools and complete multi-step financial workflows.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":["Tool use"],"topics":["Agentic","Finance","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/MiniMaxAI/MiniMax-M3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9dcbfe996a77cf11"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/bankertoolbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bankertoolbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"bankerToolBench","url":"https://benchlm.ai/benchmarks/bankertoolbench","paperUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","year":"2026","fullName":"BankerToolBench","format":"Task success rate","tasks":"Finance and banking tool-use tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"bankertoolbench","url":"https://llm-stats.com/benchmarks/bankertoolbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","finance","agents","tool calling"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"specific"},{"id":"bm_basketballbench_30421d6c","familyId":"bmf_af32ef1acc09","name":"BasketballBench","oneLine":"Evaluates basketball understanding through 7,980 questions over text, image, and video, covering ten tasks from event recognition to structured game knowledge.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.23435v1","pdf":"https://arxiv.org/pdf/2608.23435v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce BasketballBench, a multimodal benchmark comprising 7,980 questions across ten tasks in text, image, and video.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23435"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates basketball understanding through 7,980 questions over text, image, and video, covering ten tasks from event recognition to structured game knowledge.","whyItMatters":"Measures integrated multimodal reasoning in a domain where separate capabilities must work together, revealing gaps in current MLLMs.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"37678929f95431fd8f7048da069393caefe60391f56a074816a1cc95797c572e"},"motivation":"Understanding a basketball game requires recognizing events, localizing actions, identifying players, and relating these to structured game knowledge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with a defined task set and scoring, but no direct artifact links; source statement implies released data by describing its composition.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce BasketballBench, a multimodal benchmark comprising 7,980 questions across ten tasks"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.23435v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":45,"confidence":"Low","horizon":"7d","reason":"Domain-specific multimodal benchmark with no provided code or dataset links in the input, limiting immediate discoverability."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_bavground_960d10f0","familyId":"bmf_dedb900e4843","name":"BavGround","oneLine":"BavGround evaluates regional cultural grounding and dialect competence in Bavarian across English, German, and Bavarian, with 618 multi-parallel multiple-choice questions across eight cultural domains.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12894","pdf":"https://arxiv.org/pdf/2608.12894","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12894"},"evidence":{"snippet":"We introduce BavGround, a benchmark for evaluating Bavarian regional cultural grounding and dialect competence across English, German and Bavarian.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12894"},"ranking":{"30d":{"score":39,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"BavGround evaluates regional cultural grounding and dialect competence in Bavarian across English, German, and Bavarian, with 618 multi-parallel multiple-choice questions across eight cultural domains.","whyItMatters":"Addresses the gap in cultural evaluation for regional and dialect communities, providing a protocol-aware benchmark that highlights performance differences across evaluation methods and supports localized assessment of LLMs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"170f8b763b731772f208a0399c8a4fb75882b25af7d60e0e4633e969df0e9322"},"motivation":"Cultural evaluation of large language models (LLMs) often focuses on high-resource standard languages, leaving regional culture and dialect communities underrepresented.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12894","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e66f6bfdea070623","familyId":"catalog_family_e66f6bfdea070623","name":"BBH","oneLine":"Big-Bench Hard (BBH) is a suite of 23 challenging tasks selected from BIG-Bench for which prior language model evaluations did not outperform the average human-rater. These tasks require multi-step reasoning across diverse domains including arithmetic, logical reasoning, reading comprehension, and commonsense reasoning. The benchmark was designed to test capabilities believed to be beyond current language models and focuses on evaluating complex reasoning skills including temporal understanding, spatial reasoning, causal understanding, and deductive logical reasoning.","description":"Big-Bench Hard (BBH) is a suite of 23 challenging tasks selected from BIG-Bench for which prior language model evaluations did not outperform the average human-rater. These tasks require multi-step reasoning across diverse domains including arithmetic, logical reasoning, reading comprehension, and commonsense reasoning. The benchmark was designed to test capabilities believed to be beyond current language models and focuses on evaluating complex reasoning skills including temporal understanding, spatial reasoning, causal understanding, and deductive logical reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Language","Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2210.09261","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e66f6bfdea070623"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/bbh"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bbh"}],"catalogSources":[{"catalog":"benchlm","sourceId":"bbh","url":"https://benchlm.ai/benchmarks/bbh","paperUrl":"https://arxiv.org/abs/2210.09261","year":"2022","fullName":"BIG-Bench Hard","format":"Mixed reasoning tasks","tasks":"23 tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"bbh","url":"https://llm-stats.com/benchmarks/bbh","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","language","math"],"catalogModelCount":12,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_9019868a487d165b","familyId":"catalog_family_9019868a487d165b","name":"BC-VL","oneLine":"BC-VL is a vision-language benchmark for knowledge-grounded multimodal question answering.","description":"BC-VL is a vision-language benchmark for knowledge-grounded multimodal question answering.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/bc-vl","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9019868a487d165b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bc-vl"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"bc-vl","url":"https://llm-stats.com/benchmarks/bc-vl","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","multimodal","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_7bcc639830899dd7","familyId":"catalog_family_7bcc639830899dd7","name":"Beam 128K","oneLine":"Beam 128K evaluates reasoning over long inputs at a 128K-token context length.","description":"Beam 128K evaluates reasoning over long inputs at a 128K-token context length.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/beam-128k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7bcc639830899dd7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/beam-128k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"beam-128k","url":"https://llm-stats.com/benchmarks/beam-128k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_bear-bench_6aff8566","familyId":"bmf_a95253d365a4","name":"BEAR-Bench","oneLine":"Tests English–Russian multimodal reasoning over text-dense business and scientific document pages.","area":"Multimodal","applicationDomains":["Finance & Economics","Science & Research"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":["Professional document reasoning","Multimodal reasoning","OCR-grounded reasoning","Bilingual reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.17895","pdf":"https://arxiv.org/pdf/2608.17895","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.17895"},"evidence":{"snippet":"To address these limitations, we introduce BEAR-Bench (Bilingual Enterprise and Academic Reasoning), a self-contained, complex English-and-Russian benchmark comprising 1000 human-annotated questions based on text-rich business and scientific documents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17895"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"BEAR-Bench evaluates multimodal models on bilingual English-Russian text-dense business and scientific documents with 1,000 human-annotated questions.","whyItMatters":"Existing multimodal benchmarks underrepresent bilingual professional document reasoning; BEAR-Bench targets this gap for enterprise and academic use cases.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5ca9fb0fd596963ea1c90fe1ad7ee350039a840a6b14ae7ca9163a5e7e44f457"},"motivation":"While Multimodal Large Language Models (MLLMs) have made significant strides in visual comprehension, their ability to reason about text-dense, professional documents remains incompletely evaluated.","constructionDetail":"BEAR-Bench describes bilingual question answering over business and scientific document pages, but no official benchmark artifact has been released yet.","detail":{"taskBreakdown":["Business","Science"],"protocol":{"tasks":"1,000 document-image questions","primaryMetric":"Accuracy using a semantic-equivalence judge","language":"English and Russian","version":"v1"}},"curation":{"state":"source-reviewed","reviewedAt":"2026-08-20","sources":["https://arxiv.org/abs/2608.17895","https://arxiv.org/html/2608.17895v1"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17895","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"score_submission","availability":{"paperStatus":"available","githubStatus":"not_found","hfDatasetStatus":"not_found","evaluatorStatus":"paper_spec_only","submissionStatus":"not_found"},"displayEligible":false,"capabilityGroups":["Multimodal Perception"],"domainScope":"cross-domain"},{"id":"bm_behavior2trip_b70cb42e","familyId":"bmf_dc1bc5a108ad","name":"Behavior2Trip","oneLine":"Travel planning agents assist users in generating personalized travel plans by modeling their individual preferences.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.26807","pdf":"https://arxiv.org/pdf/2608.26807","project":null,"code":"https://github.com/BUAA-IRIP-LLM/Behavior2Trip","data":null,"hfPaper":"https://huggingface.co/papers/2608.26807"},"evidence":{"snippet":"To facilitate research on this task, we introduce Behavior2Trip, a benchmark constructed from one of the largest Chinese online travel platforms, comprising 11,400 instances.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26807"},"ranking":{"30d":{"score":23,"rank":108,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":312,"coverage":0.55,"confidence":"Low"}},"motivation":"Travel planning agents assist users in generating personalized travel plans by modeling their individual preferences.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 Findings","evidence":"Accepted by EMNLP 2026 Findings","evidenceUrl":"https://arxiv.org/abs/2608.26807","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"EMNLP 2026 Findings","reviewStatus":"accepted","decisionRaw":"Accepted by EMNLP 2026 Findings","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.26807","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by EMNLP 2026 Findings","level":"author-claim"}]}],"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_behaviorbench_01c04f85","familyId":"bmf_2d286f1be0d2","name":"BehaviorBench","oneLine":"Evaluates foundation models on four behavioral science capabilities: behavior prediction and simulation, strategic decision-making, subject-trait inference, and behavioral knowledge application, assessing both individual-level and distributional alignment.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24162","pdf":"https://arxiv.org/pdf/2606.24162","project":"https://umich-foreseer.github.io/behaviorbench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.24162"},"evidence":{"snippet":"We introduce BehaviorBench, a comprehensive benchmark that evaluates foundation models along four core capabilities: (1) behavior prediction and simulation, (2) strategic decision-making, (3) subject-trait inference, and (4) behavioral knowledge application.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24162"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates foundation models on four behavioral science capabilities: behavior prediction and simulation, strategic decision-making, subject-trait inference, and behavioral knowledge application, assessing both individual-level and distributional alignment.","whyItMatters":"Addresses the lack of systematic evaluation of foundation models in behavioral science, providing a standardized benchmark that captures population-level validity, helping researchers and practitioners choose models for behavioral simulation and analysis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d6eff0c3c37a7bb9add1c0946432270af6262242cf25f20be88b1394976141e9"},"motivation":"Foundation models have been increasingly applied to behavioral science domains such as psychology, sociology, and economics.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24162","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"University of Michigan Foreseer Lab","organizationType":"academic-lab","sourceUrl":"https://umich-foreseer.github.io/behaviorbench/","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_behaviorbench_fe357f72","familyId":"bmf_2d286f1be0d2","name":"BehaviorBench","oneLine":"BehaviorBench evaluates personalized decision modeling from real-world behavioral traces, with belief and trade prediction tasks from prediction-market and on-chain records across 2,000 wallets.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02798","pdf":"https://arxiv.org/pdf/2606.02798","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02798"},"evidence":{"snippet":"We introduce \\textsc{BehaviorBench}, a benchmark for evaluating personalized decision modeling from real-world behavioral traces.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02798"},"ranking":{},"description":"BehaviorBench evaluates personalized decision modeling from real-world behavioral traces, with belief and trade prediction tasks from prediction-market and on-chain records across 2,000 wallets.","whyItMatters":"Provides a real-world alternative to simulated user benchmarks, testing whether personalization methods can use observed behavioral evidence effectively in decision-support settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6a9673298d210b2353335100da0e3da5db9c69a2e0806c42a7210fb2617580f5"},"motivation":"Many decision-support settings require systems that adapt to individual users, but evaluation data for this problem remain limited.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02798","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_bekchiai-measuring-observing-and-controlli_e45a38cc","familyId":"bmf_1c2d70d8e210","name":"BekchiAI-Benchmark","oneLine":"BekchiAI-Benchmark evaluates LLM agent skills using 2,057 deterministic tool-using ReAct tasks across 7 categories, scoring accuracy plus tool-call adherence, URL hallucination, source-match, and token cost.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.26867","pdf":"https://arxiv.org/pdf/2608.26867","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present BekchiAI, which addresses both sides: a benchmark for measuring agentic skill and a platform for observing and controlling live agents.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":null,"hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26867"},"ranking":{},"description":"BekchiAI-Benchmark evaluates LLM agent skills using 2,057 deterministic tool-using ReAct tasks across 7 categories, scoring accuracy plus tool-call adherence, URL hallucination, source-match, and token cost.","whyItMatters":"Accuracy-only leaderboards miss agent-specific failures; this benchmark adds behavioral metrics and verifier-checked answers, giving teams a repeatable way to compare and debug multi-step agent behavior.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T07:23:36.749451Z","inputHash":"13d184e9f41864e1e4c815b918807cdf7b7b7951f80dde4f31254174a4c60aaf"},"motivation":"Large language model agents reason, call tools, and act autonomously over many steps, but their agentic skills-correctly sequencing tools, planning under dependencies, judging untrusted inputs, and grounding generated arguments-are hard to measure with accuracy-only leaderboards.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T07:23:36.749451Z","model":"deepseek-v4-pro","decisionReason":"The abstract declares a formal benchmark suite with explicit task counts, evaluation metrics, and public release of benchmarks, tools, and platform, satisfying a stable scoring contract and public reuse path.","canonicalNameSource":"abstract","canonicalNameEvidence":"The BekchiAI-Benchmark, a suite of 13 tool-using ReAct agents across 7 task categories"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.26867","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-30T06:59:55.242506Z"},"attentionForecast":{"score":55,"confidence":"Low","horizon":"7d","reason":"A niche agentic-skill benchmark with public tools and a four-model comparison may attract moderate interest from researchers focused on LLM agent evaluation, but lacks broad topical appeal or immediate artifact links."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_benchbench-protocol_1d14c082","familyId":"bmf_bee0f754351c","name":"BenchBench-Protocol","oneLine":"Evaluates LLMs on 149 protocol-modification tasks reconstructed from real changes scientists made to published wet-lab protocols, using weighted rubric scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-27","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.23898","pdf":"https://arxiv.org/pdf/2608.23898","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.23898"},"evidence":{"snippet":"We introduce BenchBench-Protocol, a benchmark for large language models of 149 protocol-modification tasks recovered from modifications that scientists made to published protocols during real experimental work.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23898"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLMs on 149 protocol-modification tasks reconstructed from real changes scientists made to published wet-lab protocols, using weighted rubric scoring.","whyItMatters":"Assesses routine wet-lab adaptation reasoning from real experimental modifications, filling a gap in grounded life-science evaluation beyond expert-elicited tasks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"1fa85de3c2f40f379a057f95ae025b309c06de0a2e2d6f877c66fc53087801b9"},"motivation":"We introduce BenchBench-Protocol, a benchmark for large language models of 149 protocol-modification tasks recovered from modifications that scientists made to published protocols during real experimental work.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.23898","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"attentionForecast":{"score":35,"confidence":"Low","horizon":"7d","reason":"The novelty of real-world protocol modification tasks and life-science relevance may attract moderate attention, but lack of public artifacts and a preprint-only source temper the forecast."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_fdc3093e1cbc6340","familyId":"catalog_family_fdc3093e1cbc6340","name":"BenchCAD","oneLine":"BenchCAD is a benchmark for programmatic CAD reasoning built from 17,900 execution-verified CadQuery programs spanning 106 industrial part families, roughly half anchored to real ISO, DIN, EN, ASME, and IEC specification tables. It decomposes CAD capability into matched tasks; the Vision2Code task requires models to generate CadQuery code from multi-view renders, scored by voxel IoU against the reference geometry.","description":"BenchCAD is a benchmark for programmatic CAD reasoning built from 17,900 execution-verified CadQuery programs spanning 106 industrial part families, roughly half anchored to real ISO, DIN, EN, ASME, and IEC specification tables. It decomposes CAD capability into matched tasks; the Vision2Code task requires models to generate CadQuery code from multi-view renders, scored by voxel IoU against the reference geometry.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Code","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/benchcad","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fdc3093e1cbc6340"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/benchcad"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"benchcad","url":"https://llm-stats.com/benchmarks/benchcad","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","code","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_a94b219f9fd71820","familyId":"catalog_family_a94b219f9fd71820","name":"BenchCAD (with Python tool)","oneLine":"BenchCAD variant evaluated with access to a Python tool for programmatic CAD reasoning.","description":"BenchCAD variant evaluated with access to a Python tool for programmatic CAD reasoning.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Code","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/benchcad-with-python-tool","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a94b219f9fd71820"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/benchcad-with-python-tool"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"benchcad-with-python-tool","url":"https://llm-stats.com/benchmarks/benchcad-with-python-tool","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","code","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_55ded723d335f2cd","familyId":"catalog_family_55ded723d335f2cd","name":"BenchCAD Vision2Code (no tools)","oneLine":"Generates CadQuery code from multi-view renders and scores geometric similarity by voxel intersection-over-union.","description":"Generates CadQuery code from multi-view renders and scores geometric similarity by voxel intersection-over-union.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2605.10865","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_55ded723d335f2cd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/benchcadvision2code"}],"catalogSources":[{"catalog":"benchlm","sourceId":"benchCadVision2Code","url":"https://benchlm.ai/benchmarks/benchcadvision2code","paperUrl":"https://arxiv.org/abs/2605.10865","year":"2026","fullName":"BenchCAD Vision2Code voxel IoU without tools","format":"Voxel IoU","tasks":"1,000-file Vision2Code subset","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_998a56f50d639641","familyId":"catalog_family_998a56f50d639641","name":"BenchCAD Vision2Code (tools)","oneLine":"Generates CadQuery code from multi-view renders with image inspection and code-execution tools.","description":"Generates CadQuery code from multi-view renders with image inspection and code-execution tools.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2605.10865","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_998a56f50d639641"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/benchcadvision2codewithtools"}],"catalogSources":[{"catalog":"benchlm","sourceId":"benchCadVision2CodeWithTools","url":"https://benchlm.ai/benchmarks/benchcadvision2codewithtools","paperUrl":"https://arxiv.org/abs/2605.10865","year":"2026","fullName":"BenchCAD Vision2Code voxel IoU with tools","format":"Voxel IoU with tools","tasks":"1,000-file Vision2Code subset","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_bensyc_50720d15","familyId":"bmf_7197a91c5412","name":"BenSyc","oneLine":"BenSyc is a benchmark for studying conversational sycophancy in Bengali social contexts, with binary labels and a five-level taxonomy from invalidation to escalation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.10061","pdf":"https://arxiv.org/pdf/2606.10061","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.10061"},"evidence":{"snippet":"We introduce BenSyc, the first benchmark for studying conversational sycophancy in Bengali social contexts.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":"2026-06-10T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.10061"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"BenSyc is a benchmark for studying conversational sycophancy in Bengali social contexts, with binary labels and a five-level taxonomy from invalidation to escalation.","whyItMatters":"It fills a gap in sycophancy research by focusing on culturally grounded conversational alignment in Bengali, important for socially aligned conversational AI.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3876527e6a234d1ab3875db4479c7c08bd5527343dec92d674037da17c7f8e0f"},"motivation":"Large language models (LLMs) increasingly participate in emotionally sensitive social conversations, where responses may shift from balanced support toward excessive validation or escalatory alignment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.10061","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"lib_bfcl","familyId":"family_bfcl","name":"Berkeley Function-Calling Leaderboard","oneLine":"Established benchmark family · Agents & Tool Use.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents & Tool Use"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2402.18679","pdf":null,"project":"https://gorilla.cs.berkeley.edu/leaderboard.html","code":"https://github.com/ShishirPatil/gorilla","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_bfcl"},"ranking":{},"recordType":"family","aliases":["BFCL"],"sourceAttribution":[{"role":"official-leaderboard","url":"https://gorilla.cs.berkeley.edu/leaderboard.html"}],"adoptionRefs":["openai-gpt5","google-gemini25"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"bfcl","url":"https://llm-stats.com/benchmarks/bfcl","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","tool calling"],"catalogModelCount":11,"catalogStarCount":0},{"id":"catalog_ccaa76994404b451","familyId":"catalog_family_ccaa76994404b451","name":"Beyond AIME","oneLine":"Beyond AIME is a difficult mathematical reasoning benchmark designed to test deeper reasoning chains and harder decomposition than standard AIME-style problem sets.","description":"Beyond AIME is a difficult mathematical reasoning benchmark designed to test deeper reasoning chains and harder decomposition than standard AIME-style problem sets.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/beyond-aime","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ccaa76994404b451"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/beyond-aime"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"beyond-aime","url":"https://llm-stats.com/benchmarks/beyond-aime","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_beyondmasks_787c429f","familyId":"bmf_159cf4e8d619","name":"BeyondMasks","oneLine":"Evaluates causally consistent video object removal using paired synthetic and real videos with aligned references, and includes the CORE metric for after-effect consistency.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.20107","pdf":"https://arxiv.org/pdf/2608.20107","project":"https://yigitekin.github.io/BeyondMasks/","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce BeyondMasks, a paired benchmark for causally consistent video object removal, consisting of temporally aligned synthetic and real world video pairs with clean background references.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.20107"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates causally consistent video object removal using paired synthetic and real videos with aligned references, and includes the CORE metric for after-effect consistency.","whyItMatters":"Introduces a benchmark that checks whether removed objects' induced physical effects are also eliminated, addressing a key gap in video editing evaluation beyond masked region fidelity.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"4cc797982d43ab541b6f89cd22ec5b63f1e1309acfd0c9111b3afaba58ae178a"},"motivation":"Recent advances in generative video models have significantly improved visual realism in video object removal, yet evaluation protocols still focus on masked region fidelity, treating removal as local inpainting.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"The Project Page is provided and the paper defines a paired dataset with references and an evaluation protocol, enabling reuse by other teams.","canonicalNameSource":"paper_title","canonicalNameEvidence":"BeyondMasks: Evaluating Causal and Physical Consistency in Video Object Removal"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.20107","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:50.392548Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"The ECCV venue and project page link increase visibility, and causal consistency evaluation is an emerging area likely to draw interest."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_5c197b338806921e","familyId":"catalog_family_5c197b338806921e","name":"BFCL v2","oneLine":"Berkeley Function Calling Leaderboard (BFCL) v2 is a comprehensive benchmark for evaluating large language models' function calling capabilities. It features 2,251 question-function-answer pairs with enterprise and OSS-contributed functions, addressing data contamination and bias through live, user-contributed scenarios. The benchmark evaluates AST accuracy, executable accuracy, irrelevance detection, and relevance detection across multiple programming languages (Python, Java, JavaScript) and includes complex real-world function calling scenarios with multi-lingual prompts.","description":"Berkeley Function Calling Leaderboard (BFCL) v2 is a comprehensive benchmark for evaluating large language models' function calling capabilities. It features 2,251 question-function-answer pairs with enterprise and OSS-contributed functions, addressing data contamination and bias through live, user-contributed scenarios. The benchmark evaluates AST accuracy, executable accuracy, irrelevance detection, and relevance detection across multiple programming languages (Python, Java, JavaScript) and includes complex real-world function calling scenarios with multi-lingual prompts.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","General","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/bfcl-v2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5c197b338806921e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bfcl-v2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"bfcl-v2","url":"https://llm-stats.com/benchmarks/bfcl-v2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","tool calling"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_072cecd64f67d822","familyId":"catalog_family_072cecd64f67d822","name":"BFCL v4","oneLine":"Berkeley Function Calling Leaderboard V4 (BFCL-V4) evaluates LLMs on their ability to accurately call functions and APIs, including simple, multiple, parallel, and nested function calls across diverse programming scenarios.","description":"Berkeley Function Calling Leaderboard V4 (BFCL-V4) evaluates LLMs on their ability to accurately call functions and APIs, including simple, multiple, parallel, and nested function calls across diverse programming scenarios.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agentic","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.arcee.ai/blog/trinity-large-thinking","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_072cecd64f67d822"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/bfcl-v4"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bfcl-v4"}],"catalogSources":[{"catalog":"benchlm","sourceId":"bfclV4","url":"https://benchlm.ai/benchmarks/bfcl-v4","paperUrl":"https://www.arcee.ai/blog/trinity-large-thinking","year":"2026","fullName":"Berkeley Function Calling Leaderboard v4","format":"Tool invocation and schema evaluation","tasks":"Function-calling tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"bfcl-v4","url":"https://llm-stats.com/benchmarks/bfcl-v4","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","agents","tool calling"],"catalogModelCount":15,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_d0ca5bf6b14c4f58","familyId":"catalog_family_d0ca5bf6b14c4f58","name":"BFCL-v3","oneLine":"Berkeley Function Calling Leaderboard v3 (BFCL-v3) is an advanced benchmark that evaluates large language models' function calling capabilities through multi-turn and multi-step interactions. It introduces extended conversational exchanges where models must retain contextual information across turns and execute multiple internal function calls for complex user requests. The benchmark includes 1000 test cases across domains like vehicle control, trading bots, travel booking, and file system management, using state-based evaluation to verify both system state changes and execution path correctness.","description":"Berkeley Function Calling Leaderboard v3 (BFCL-v3) is an advanced benchmark that evaluates large language models' function calling capabilities through multi-turn and multi-step interactions. It introduces extended conversational exchanges where models must retain contextual information across turns and execute multiple internal function calls for complex user requests. The benchmark includes 1000 test cases across domains like vehicle control, trading bots, travel booking, and file system management, using state-based evaluation to verify both system state changes and execution path correctness.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Structured Output","Finance","General","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/bfcl-v3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d0ca5bf6b14c4f58"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bfcl-v3"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"bfcl-v3","url":"https://llm-stats.com/benchmarks/bfcl-v3","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","structured output","finance","general","agents","tool calling"],"catalogModelCount":19,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"specific"},{"id":"catalog_30e8ebdf60c3b1f9","familyId":"catalog_family_30e8ebdf60c3b1f9","name":"BFCL_v3_MultiTurn","oneLine":"Berkeley Function Calling Leaderboard (BFCL) V3 MultiTurn benchmark that evaluates large language models' ability to handle multi-turn and multi-step function calling scenarios. The benchmark introduces complex interactions requiring models to manage sequential function calls, handle conversational context across multiple turns, and make dynamic decisions about when and how to use available functions. BFCL V3 uses state-based evaluation by verifying the actual state of API systems after function execution, providing more realistic assessment of function calling capabilities in agentic applications.","description":"Berkeley Function Calling Leaderboard (BFCL) V3 MultiTurn benchmark that evaluates large language models' ability to handle multi-turn and multi-step function calling scenarios. The benchmark introduces complex interactions requiring models to manage sequential function calls, handle conversational context across multiple turns, and make dynamic decisions about when and how to use available functions. BFCL V3 uses state-based evaluation by verifying the actual state of API systems after function execution, providing more realistic assessment of function calling capabilities in agentic applications.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","General","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/bfcl-v3-multiturn","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_30e8ebdf60c3b1f9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bfcl-v3-multiturn"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"bfcl-v3-multiturn","url":"https://llm-stats.com/benchmarks/bfcl-v3-multiturn","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","tool calling"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_bg-real_ba34eeb5","familyId":"bmf_6a40ec51857c","name":"BG-REAL","oneLine":"BG-REAL is a benchmark for background manipulation detection and localization in images. It contains 7,000 processed samples (6,000 public-data anchored, 1,000 synthetic) over six edit families with matched authentic controls, source-group splits, and quality control.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.26232","pdf":"https://arxiv.org/pdf/2607.26232","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.26232"},"evidence":{"snippet":"We introduce BG-REAL, a public real-data anchored benchmark package for background manipulation detection and localization.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26232"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"BG-REAL is a benchmark for background manipulation detection and localization in images. It contains 7,000 processed samples (6,000 public-data anchored, 1,000 synthetic) over six edit families with matched authentic controls, source-group splits, and quality control.","whyItMatters":"Existing image forensics benchmarks focus on object-centric manipulations, missing background edits. BG-REAL provides a targeted evaluation with matched controls to measure false-positive rates from re-encoding artifacts, exposing a shared shortcut risk across baselines.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"95c49a9214a32e9c3fd941275aa030eb8a7c6ddf2a5eac0d3877fc40bf785bc1"},"motivation":"Background manipulation is a practical but under-specified image-forensics setting: the manipulated evidence can sit outside the salient foreground object, while many evaluations emphasize object-centric copy-move, splicing, or generic synthetic edits.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26232","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_efc9be1bb3b5bc96","familyId":"catalog_family_efc9be1bb3b5bc96","name":"Big Bench Audio","oneLine":"Big Bench Audio is an audio reasoning benchmark adapted from a subset of Big Bench Hard, with text questions converted to spoken audio. It evaluates the reasoning ability of speech-to-speech and audio language models on tasks delivered as audio input, with accuracy scored by an independent evaluation (Artificial Analysis).","description":"Big Bench Audio is an audio reasoning benchmark adapted from a subset of Big Bench Hard, with text questions converted to spoken audio. It evaluates the reasoning ability of speech-to-speech and audio language models on tasks delivered as audio input, with accuracy scored by an independent evaluation (Artificial Analysis).","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/big-bench-audio","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_efc9be1bb3b5bc96"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/big-bench-audio"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"big-bench-audio","url":"https://llm-stats.com/benchmarks/big-bench-audio","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","audio"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_c240e59be7f89744","familyId":"catalog_family_c240e59be7f89744","name":"BIG-Bench","oneLine":"Beyond the Imitation Game Benchmark (BIG-bench) is a collaborative benchmark consisting of 204+ tasks designed to probe large language models and extrapolate their future capabilities. It covers diverse domains including linguistics, mathematics, common-sense reasoning, biology, physics, social bias, software development, and more. The benchmark focuses on tasks believed to be beyond current language model capabilities and includes both English and non-English tasks across multiple languages.","description":"Beyond the Imitation Game Benchmark (BIG-bench) is a collaborative benchmark consisting of 204+ tasks designed to probe large language models and extrapolate their future capabilities. It covers diverse domains including linguistics, mathematics, common-sense reasoning, biology, physics, social bias, software development, and more. The benchmark focuses on tasks believed to be beyond current language model capabilities and includes both English and non-English tasks across multiple languages.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/big-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c240e59be7f89744"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/big-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"big-bench","url":"https://llm-stats.com/benchmarks/big-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","math","reasoning"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_c4b9e72d07024860","familyId":"catalog_family_c4b9e72d07024860","name":"BIG-Bench Extra Hard","oneLine":"BIG-Bench Extra Hard (BBEH) is a challenging benchmark that replaces each task in BIG-Bench Hard with a novel task that probes similar reasoning capabilities but exhibits significantly increased difficulty. The benchmark contains 23 tasks testing diverse reasoning skills including many-hop reasoning, causal understanding, spatial reasoning, temporal arithmetic, geometric reasoning, linguistic reasoning, logic puzzles, and humor understanding. Designed to address saturation on existing benchmarks where state-of-the-art models achieve near-perfect scores, BBEH shows substantial room for improvement with best models achieving only 9.8-44.8% average accuracy.","description":"BIG-Bench Extra Hard (BBEH) is a challenging benchmark that replaces each task in BIG-Bench Hard with a novel task that probes similar reasoning capabilities but exhibits significantly increased difficulty. The benchmark contains 23 tasks testing diverse reasoning skills including many-hop reasoning, causal understanding, spatial reasoning, temporal arithmetic, geometric reasoning, linguistic reasoning, logic puzzles, and humor understanding. Designed to address saturation on existing benchmarks where state-of-the-art models achieve near-perfect scores, BBEH shows substantial room for improvement with best models achieving only 9.8-44.8% average accuracy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/big-bench-extra-hard","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c4b9e72d07024860"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/big-bench-extra-hard"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"big-bench-extra-hard","url":"https://llm-stats.com/benchmarks/big-bench-extra-hard","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning","general"],"catalogModelCount":11,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_ad505a3aa6e37229","familyId":"catalog_family_ad505a3aa6e37229","name":"BIG-Bench Hard","oneLine":"BIG-Bench Hard (BBH) is a subset of 23 challenging BIG-Bench tasks selected because prior language model evaluations did not outperform average human-rater performance. The benchmark contains 6,511 evaluation examples testing various forms of multi-step reasoning including arithmetic, logical reasoning (Boolean expressions, logical deduction), geometric reasoning, temporal reasoning, and language understanding. Tasks require capabilities such as causal judgment, object counting, navigation, pattern recognition, and complex problem solving.","description":"BIG-Bench Hard (BBH) is a subset of 23 challenging BIG-Bench tasks selected because prior language model evaluations did not outperform average human-rater performance. The benchmark contains 6,511 evaluation examples testing various forms of multi-step reasoning including arithmetic, logical reasoning (Boolean expressions, logical deduction), geometric reasoning, temporal reasoning, and language understanding. Tasks require capabilities such as causal judgment, object counting, navigation, pattern recognition, and complex problem solving.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/big-bench-hard","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ad505a3aa6e37229"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/big-bench-hard"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"big-bench-hard","url":"https://llm-stats.com/benchmarks/big-bench-hard","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","math","reasoning"],"catalogModelCount":21,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_de0a9cde8ad3201c","familyId":"catalog_family_de0a9cde8ad3201c","name":"BigCodeBench","oneLine":"A benchmark that challenges LLMs to invoke multiple function calls as tools from 139 libraries and 7 domains for 1,140 fine-grained programming tasks. Evaluates code generation with diverse function calls and complex instructions, featuring two variants: Complete (code completion based on comprehensive docstrings) and Instruct (generating code from natural language instructions).","description":"A benchmark that challenges LLMs to invoke multiple function calls as tools from 139 libraries and 7 domains for 1,140 fine-grained programming tasks. Evaluates code generation with diverse function calls and complex instructions, featuring two variants: Complete (code completion based on comprehensive docstrings) and Instruct (generating code from natural language instructions).","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_de0a9cde8ad3201c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/bigcodebench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bigcodebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"bigCodeBench","url":"https://benchlm.ai/benchmarks/bigcodebench","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"BigCodeBench","format":"Pass@1","tasks":"Code generation tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"bigcodebench","url":"https://llm-stats.com/benchmarks/bigcodebench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","reasoning","general"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_3d7c9b2f8573753d","familyId":"catalog_family_3d7c9b2f8573753d","name":"BigCodeBench-Full","oneLine":"A comprehensive benchmark that evaluates large language models' ability to solve complex, practical programming tasks via code generation. Contains 1,140 fine-grained tasks across 7 domains using function calls from 139 libraries. Challenges LLMs to invoke multiple function calls as tools and handle complex instructions for realistic software engineering and general-purpose reasoning tasks.","description":"A comprehensive benchmark that evaluates large language models' ability to solve complex, practical programming tasks via code generation. Contains 1,140 fine-grained tasks across 7 domains using function calls from 139 libraries. Challenges LLMs to invoke multiple function calls as tools and handle complex instructions for realistic software engineering and general-purpose reasoning tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/bigcodebench-full","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3d7c9b2f8573753d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bigcodebench-full"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"bigcodebench-full","url":"https://llm-stats.com/benchmarks/bigcodebench-full","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d95852dd185a1bb1","familyId":"catalog_family_d95852dd185a1bb1","name":"BigCodeBench-Hard","oneLine":"BigCodeBench-Hard is a subset of 148 challenging programming tasks from BigCodeBench, designed to evaluate large language models' ability to solve complex, real-world programming problems. These tasks require diverse function calls from multiple libraries across 7 domains including computation, networking, data analysis, and visualization. The benchmark tests compositional reasoning and the ability to implement complex instructions that span 139 libraries with an average of 2.8 libraries per task.","description":"BigCodeBench-Hard is a subset of 148 challenging programming tasks from BigCodeBench, designed to evaluate large language models' ability to solve complex, real-world programming problems. These tasks require diverse function calls from multiple libraries across 7 domains including computation, networking, data analysis, and visualization. The benchmark tests compositional reasoning and the ability to implement complex instructions that span 139 libraries with an average of 2.8 libraries per task.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/bigcodebench-hard","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d95852dd185a1bb1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bigcodebench-hard"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"bigcodebench-hard","url":"https://llm-stats.com/benchmarks/bigcodebench-hard","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_bigfinancebench_c8640763","familyId":"bmf_4b9b21faaf9f","name":"BigFinanceBench","oneLine":"BigFinanceBench evaluates financial-research agents on open-ended tasks with 928 items, each paired with a ground-truth answer and a point-weighted rubric decomposing the derivation into steps. Supports partial-credit scoring across 36,241 rubric points.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03829","pdf":"https://arxiv.org/pdf/2606.03829","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03829"},"evidence":{"snippet":"We introduce BigFinanceBench, a 928-item expert-authored benchmark of open-ended financial-research tasks in which each item pairs a ground-truth reference answer with a point-weighted rubric that decomposes the derivation into independently checkable steps.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03829"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"BigFinanceBench evaluates financial-research agents on open-ended tasks with 928 items, each paired with a ground-truth answer and a point-weighted rubric decomposing the derivation into steps. Supports partial-credit scoring across 36,241 rubric points.","whyItMatters":"Existing finance benchmarks evaluate subskills or final answers, not the auditable derivation. BigFinanceBench measures workflow quality, allowing localization of failures and better assessment of decision-relevant outputs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"883397bd70250ab65f8b39beec3514a8acdf14a941036f77a5cfae507da703d9"},"motivation":"Financial-research answers are decision-relevant only when another analyst can audit how they were produced: which source was chosen, which period and accounting definition were used, which assumptions were made, and how the calculation was performed.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03829","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific","catalogSources":[{"catalog":"llm-stats","sourceId":"big-finance-bench","url":"https://llm-stats.com/benchmarks/big-finance-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","finance","agents"],"catalogModelCount":3,"catalogStarCount":0},{"id":"bm_billiardphys-bench_142a32a2","familyId":"bmf_750da7a3f066","name":"BilliardPhys-Bench","oneLine":"Evaluates physical reasoning in synthetic billiards environments, testing collision prediction, wall bounce reasoning, and final position estimation for multimodal LLMs.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.30900","pdf":"https://arxiv.org/pdf/2605.30900","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30900"},"evidence":{"snippet":"We present BilliardPhys-Bench, a benchmark for physical reasoning in synthetic billiards environments.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30900"},"ranking":{},"description":"Evaluates physical reasoning in synthetic billiards environments, testing collision prediction, wall bounce reasoning, and final position estimation for multimodal LLMs.","whyItMatters":"Addresses the gap in evaluating visual dynamics and physical reasoning capabilities of multimodal models, providing a method to assess model performance on intuitive physics tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"832ac2d65f551f1a6e561fc2e727200874ee255bb30dc8fc0fe9ab615919f99a"},"motivation":"Current multimodal models handle static image recognition well, but intuitive physical reasoning remains a weakness.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30900","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_bim-edit_44064775","familyId":"bmf_2a8fff911c4c","name":"BIM-Edit","oneLine":"BIM-Edit evaluates large language models on natural-language editing of Industry Foundation Classes (IFC) building models. The benchmark includes 324 editing tasks across 11 realistic building models and 36 synthetic scenes. Tasks are categorized as direct, spatial, or topological instructions, and outputs are scored on geometric accuracy, semantic validity, and topological consistency.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.20146","pdf":"https://arxiv.org/pdf/2606.20146","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.20146"},"evidence":{"snippet":"We introduce BIM-Edit, a benchmark for evaluating LLMs on natural-language editing of Building Information Models (BIM) represented in the Industry Foundation Classes (IFC) format.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.20146"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"BIM-Edit evaluates large language models on natural-language editing of Industry Foundation Classes (IFC) building models. The benchmark includes 324 editing tasks across 11 realistic building models and 36 synthetic scenes. Tasks are categorized as direct, spatial, or topological instructions, and outputs are scored on geometric accuracy, semantic validity, and topological consistency.","whyItMatters":"Construction and architectural design rely on structured BIM models; the evaluation gap is that existing benchmarks mostly test geometry and creation from scratch. BIM-Edit measures scene understanding and semantic relational preservation, providing a capability signal for practical engineering workflows where editing is central.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a0d8c53ce91c5b0295325792708c88fde7f7f4ec7fe0c3ae31b4181e646ddc89"},"motivation":"Large language models (LLMs) are increasingly applied to computer-aided design (CAD) to generate design artifacts from textual instructions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.20146","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_binjudgebench_a8dd4ae5","familyId":"bmf_03c435b933bf","name":"BinJudgeBench","oneLine":"BinJudgeBench evaluates LLM-as-a-Judge for human-oriented binary reverse engineering, covering function name recovery, code summarization, and decompilation optimization, with correlation to human judgment as the metric.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07038","pdf":"https://arxiv.org/pdf/2608.07038","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07038"},"evidence":{"snippet":"We introduce BinJudgeBench, the first expert-annotated, reference-free evaluation benchmark based on multi-dimensional human judgment, where LLM-as-a-Judge achieves an average correlation of 63.20\\% with human judgment, outperforming traditional automated metrics at 35.04\\%.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07038"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"BinJudgeBench evaluates LLM-as-a-Judge for human-oriented binary reverse engineering, covering function name recovery, code summarization, and decompilation optimization, with correlation to human judgment as the metric.","whyItMatters":"Addresses the challenge of scalable evaluation for HOBRE, providing a reference-free alternative that correlates with human judgment better than traditional metrics.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4b4ab63edcb579c568a05f86c94df18cc102240ad5090bbf7dac85476beb9f54"},"motivation":"Human-Oriented Binary Reverse Engineering (HOBRE) aims to transform decompiled pseudocode into a more human-friendly representation, thereby reducing the cognitive burden of reverse analysis and improving efficiency.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"41st IEEE/ACM International Conference on Automated Software Engineering (ASE 2026)","evidence":"Accepted by the 41st IEEE/ACM International Conference on Automated Software Engineering (ASE 2026)","evidenceUrl":"https://arxiv.org/abs/2608.07038","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"41st IEEE/ACM International Conference on Automated Software Engineering (ASE 2026)","reviewStatus":"accepted","decisionRaw":"Accepted by the 41st IEEE/ACM International Conference on Automated Software Engineering (ASE 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.07038","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by the 41st IEEE/ACM International Conference on Automated Software Engineering (ASE 2026)","level":"author-claim"}]}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_7a6be30323cef9fb","familyId":"catalog_family_7a6be30323cef9fb","name":"BioLP-Bench","oneLine":"BioLP-Bench is a model-graded evaluation measuring ability to find and correct mistakes in common biological laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.","description":"BioLP-Bench is a model-graded evaluation measuring ability to find and correct mistakes in common biological laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Safety","Healthcare","Biology"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/biolp-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7a6be30323cef9fb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/biolp-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"biolp-bench","url":"https://llm-stats.com/benchmarks/biolp-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety","healthcare","biology"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"catalog_9c973121d43bf296","familyId":"catalog_family_9c973121d43bf296","name":"BioMysteryBench","oneLine":"BioMysteryBench evaluates a model's ability to reason through challenging molecular biology problems, reporting performance on a hard subset and on the subset of problems solved by human experts.","description":"BioMysteryBench evaluates a model's ability to reason through challenging molecular biology problems, reporting performance on a hard subset and on the subset of problems solved by human experts.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Reasoning","Science","Biology"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/biomysterybench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9c973121d43bf296"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/biomysterybench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"biomysterybench","url":"https://llm-stats.com/benchmarks/biomysterybench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","science","biology"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_109e010c1ce17e77","familyId":"catalog_family_109e010c1ce17e77","name":"BioMysteryBench (human-difficult)","oneLine":"Computational biology challenges with objective answers that remained unsolved by independent human experts.","description":"Computational biology challenges with objective answers that remained unsolved by independent human experts.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_109e010c1ce17e77"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/biomysterybenchhumandifficult"}],"catalogSources":[{"catalog":"benchlm","sourceId":"bioMysteryBenchHumanDifficult","url":"https://benchlm.ai/benchmarks/biomysterybenchhumandifficult","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"BioMysteryBench Human Difficult","format":"Task score","tasks":"Human-difficult computational biology investigations","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_5a9ad0293f81bcfe","familyId":"catalog_family_5a9ad0293f81bcfe","name":"BioMysteryBench (human-solvable)","oneLine":"Computational biology challenges that independent human experts were able to solve.","description":"Computational biology challenges that independent human experts were able to solve.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5a9ad0293f81bcfe"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/biomysterybenchhumansolvable"}],"catalogSources":[{"catalog":"benchlm","sourceId":"bioMysteryBenchHumanSolvable","url":"https://benchlm.ai/benchmarks/biomysterybenchhumansolvable","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"BioMysteryBench Human Solvable","format":"Task score","tasks":"Human-solvable computational biology investigations","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_biosecbench-refusal_d798db06","familyId":"bmf_f88a84efd022","name":"BioSecBench-Refusal","oneLine":"BioSecBench-Refusal evaluates AI agents on 61 Routine and 46 Red-Team biosecurity-related tasks, measuring refusal rates and risk identification across multiple model configurations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.05462","pdf":"https://arxiv.org/pdf/2607.05462","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.05462"},"evidence":{"snippet":"We present BioSecBench-Refusal, a benchmark for risk identification and refusal behavior for biological research tasks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.05462"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"BioSecBench-Refusal evaluates AI agents on 61 Routine and 46 Red-Team biosecurity-related tasks, measuring refusal rates and risk identification across multiple model configurations.","whyItMatters":"Addresses the need for benchmarks that quantify both capability and safety in agentic biosecurity, helping developers calibrate models to avoid over-refusal while still detecting threats.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"71677a53bbce6ee720e8803b2e8c1cf8f83cbb421da9d89c5b1b99e1e5d8bf29"},"motivation":"As AI agents are incorporated into life science workflows, the capabilities that speed discovery might also enable misuse.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.05462","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_biosecbench-surveillance_edf01c2c","familyId":"bmf_bbdaab9ec888","name":"BioSecBench-Surveillance","oneLine":"BioSecBench-Surveillance evaluates AI agents on pathogen genomic surveillance across 100 tasks spanning seven categories, including taxonomic classification and genetic-engineering detection. Agents receive raw sequencing data and surveillance context, and their structured answers are graded deterministically against ground truth.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19262","pdf":"https://arxiv.org/pdf/2607.19262","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19262"},"evidence":{"snippet":"We present BioSecBench-Surveillance, a verifiable benchmark of 100 evaluations testing whether AI agents can infer the right analysis pipeline from raw sequencing data and surveillance context.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19262"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"BioSecBench-Surveillance evaluates AI agents on pathogen genomic surveillance across 100 tasks spanning seven categories, including taxonomic classification and genetic-engineering detection. Agents receive raw sequencing data and surveillance context, and their structured answers are graded deterministically against ground truth.","whyItMatters":"The benchmark addresses the lack of verifiable evaluation for AI agents in genomic surveillance, where analysis bottlenecks are emerging as data generation scales. It provides a standardized measure of agent reliability in critical public health applications, with practical value in assessing whether agents can be trusted for real-world outbreak response.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fe37268113d08f0cb80c2eafa3c68e1fa16d460b6d012be238bd6aa9d579811b"},"motivation":"As pathogen genomic surveillance scales, the bottleneck is shifting from data generation to analysis.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19262","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_e8085090a8ed817a","familyId":"catalog_family_e8085090a8ed817a","name":"Bird-SQL (dev)","oneLine":"BIRD (BIg Bench for LaRge-scale Database Grounded Text-to-SQLs) is a comprehensive text-to-SQL benchmark containing 12,751 question-SQL pairs across 95 databases (33.4 GB total) spanning 37+ professional domains. It evaluates large language models' ability to convert natural language to executable SQL queries in real-world scenarios with complex database schemas and dirty data.","description":"BIRD (BIg Bench for LaRge-scale Database Grounded Text-to-SQLs) is a comprehensive text-to-SQL benchmark containing 12,751 question-SQL pairs across 95 databases (33.4 GB total) spanning 37+ professional domains. It evaluates large language models' ability to convert natural language to executable SQL queries in real-world scenarios with complex database schemas and dirty data.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/bird-sql-(dev)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e8085090a8ed817a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bird-sql-(dev)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"bird-sql-(dev)","url":"https://llm-stats.com/benchmarks/bird-sql-(dev)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":7,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_7175aad2c4803478","familyId":"catalog_family_7175aad2c4803478","name":"BixBench","oneLine":"BixBench is a benchmark for real-world bioinformatics and computational biology data analysis. It evaluates AI models on multi-step scientific workflows that require code execution, statistical reasoning, and biological domain knowledge to interpret experimental data.","description":"BixBench is a benchmark for real-world bioinformatics and computational biology data analysis. It evaluates AI models on multi-step scientific workflows that require code execution, statistical reasoning, and biological domain knowledge to interpret experimental data.","area":"Agents & Tool Use","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Reasoning","Science","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/bixbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7175aad2c4803478"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/bixbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"bixbench","url":"https://llm-stats.com/benchmarks/bixbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","science","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_bixbench3_ede996e4","familyId":"bmf_113a1bb4bad5","name":"BixBench3","oneLine":"Evaluates AI agents on 20 research-study-scale computational biology tasks that require processing raw data into scientific results, scored by matching generated artifacts against study-derived targets.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-26","firstSeenAt":"2026-08-29","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25286","pdf":"https://arxiv.org/pdf/2608.25286","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.25286"},"evidence":{"snippet":"Here we introduce BixBench3, a benchmark that measures the capacity of AI agents to process raw biological data through to scientific results.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25286"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates AI agents on 20 research-study-scale computational biology tasks that require processing raw data into scientific results, scored by matching generated artifacts against study-derived targets.","whyItMatters":"Systematically measures agent capability in long-horizon, data-intensive biological analyses, revealing gaps in sequential reasoning and resource management.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"0a7edf95f14cdbd800d23c121f31be45a70353fd30d4bce82cf9f4496b410fc9"},"motivation":"Artificial intelligence (AI) promises to accelerate biological research by automating computational analyses.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25286","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"attentionForecast":{"score":60,"confidence":"Low","horizon":"7d","reason":"The benchmark addresses a high-impact scientific domain and reports performance across many frontier models, but missing artifacts may reduce immediate attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_blind-spots-bench_1b74311c","familyId":"bmf_74da405a337c","name":"Blind-Spots-Bench","oneLine":"Evaluates reasoning blind spots in language, vision-language, and image-generation models across 235 samples with structured reference solutions and taxonomy.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.08317","pdf":"https://arxiv.org/pdf/2607.08317","project":null,"code":"https://github.com/matteosantelmo/reasoning-blind-spots","data":null,"hfPaper":"https://huggingface.co/papers/2607.08317"},"evidence":{"snippet":"We introduce $\\texttt{blind-spots-bench}$, a benchmark designed to expose such blind spots through tasks that appear simple for humans but remain challenging for modern AI.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":37,"hfDailySubmittedAt":"2026-07-15T00:00:00.000Z","githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.08317"},"ranking":{"90d":{"score":37,"rank":180,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates reasoning blind spots in language, vision-language, and image-generation models across 235 samples with structured reference solutions and taxonomy.","whyItMatters":"Serves as a diagnostic stress test exposing tasks that humans find easy but AI models struggle with, highlighting gaps not captured by existing benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2ee38fb7fce4dca5a602a776c08b2ff548172b2b38b3d879b249f7720491acf9"},"motivation":"Modern AI models achieve strong performance on many established benchmarks, yet they still fail on tasks that humans find almost trivial, such as manipulating a string or drawing a dog with five legs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.08317","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_e14a51fded1fb9b4","familyId":"catalog_family_e14a51fded1fb9b4","name":"BLINK","oneLine":"BLINK: Multimodal Large Language Models Can See but Not Perceive. A benchmark for multimodal language models focusing on core visual perception abilities. Reformats 14 classic computer vision tasks into 3,807 multiple-choice questions paired with single or multiple images and visual prompting. Tasks include relative depth estimation, visual correspondence, forensics detection, multi-view reasoning, counting, object localization, and spatial reasoning that humans can solve 'within a blink'.","description":"BLINK: Multimodal Large Language Models Can See but Not Perceive. A benchmark for multimodal language models focusing on core visual perception abilities. Reformats 14 classic computer vision tasks into 3,807 multiple-choice questions paired with single or multiple images and visual prompting. Tasks include relative depth estimation, visual correspondence, forensics detection, multi-view reasoning, counting, object localization, and spatial reasoning that humans can solve 'within a blink'.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Spatial Reasoning","3D","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/blink","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e14a51fded1fb9b4"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/blink"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"blink","url":"https://llm-stats.com/benchmarks/blink","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","spatial reasoning","3d","vision"],"catalogModelCount":15,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_bluefin_2fd6c32a","familyId":"bmf_98253ff66ed8","name":"BlueFin","oneLine":"Evaluates LLM agents on synthesis, manipulation, and comprehension tasks over financial spreadsheet workbooks, with 131 tasks and granular rubric criteria validated by expert annotators.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems","Finance & Economics"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics","Financial Services"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30907","pdf":"https://arxiv.org/pdf/2605.30907","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30907"},"evidence":{"snippet":"We present BlueFin, a benchmark that tasks large language model (LLM) agents with synthesis, manipulation, and comprehension tasks over spreadsheet workbooks in the professional finance domain.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30907"},"ranking":{},"description":"Evaluates LLM agents on synthesis, manipulation, and comprehension tasks over financial spreadsheet workbooks, with 131 tasks and granular rubric criteria validated by expert annotators.","whyItMatters":"Fills the gap in evaluating LLMs for spreadsheet tasks relevant to professional finance, where current models perform below 50%, providing a measure for practical deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"90d53d7dea828e6760a3774046a5fecedc8289acee860d28ed48ca340160a8ea"},"motivation":"We present BlueFin, a benchmark that tasks large language model (LLM) agents with synthesis, manipulation, and comprehension tasks over spreadsheet workbooks in the professional finance domain.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30907","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"BlueFin Benchmark Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2605.30907","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"cross-domain"},{"id":"catalog_c828085ad9338ced","familyId":"catalog_family_c828085ad9338ced","name":"Blueprint-Bench 2","oneLine":"Blueprint-Bench 2 is an agentic spatial reasoning benchmark that evaluates a model's ability to understand, plan, and reason over architectural blueprints and other structured spatial documents. Scores are reported as a normalized score.","description":"Blueprint-Bench 2 is an agentic spatial reasoning benchmark that evaluates a model's ability to understand, plan, and reason over architectural blueprints and other structured spatial documents. Scores are reported as a normalized score.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://x.com/GoogleDeepMind","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c828085ad9338ced"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/blueprintbench2"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/blueprint-bench-2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"blueprintBench2","url":"https://benchlm.ai/benchmarks/blueprintbench2","paperUrl":"https://x.com/GoogleDeepMind","year":"2026","fullName":"Blueprint-Bench 2","format":"Normalized score","tasks":"Spatial reasoning from blueprints","successorKey":null},{"catalog":"llm-stats","sourceId":"blueprint-bench-2","url":"https://llm-stats.com/benchmarks/blueprint-bench-2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_bluex-v2_e4bf2712","familyId":"bmf_c1d585d91de1","name":"BLUEX v2","oneLine":"BLUEX v2 evaluates large language models on open-ended, discursive questions from the second-phase entrance exams of UNICAMP and USP (2022–2025). The dataset includes 395 questions with 919 subquestions, covering nine subjects, with 55.7% of questions containing images represented as context-aware captions. Scoring uses an LLM-as-a-judge protocol with binary rubric criteria based on official reference answers, yielding a 0–10 score.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22723","pdf":"https://arxiv.org/pdf/2606.22723","project":null,"code":"https://github.com/TropicAI-Research/BLUEXv2","data":"https://huggingface.co/datasets/Tropic-AI/BLUEX-v2","hfPaper":"https://huggingface.co/papers/2606.22723"},"evidence":{"snippet":"In this work, we introduce BLUEX v2, a benchmark derived from the second-phase entrance exams of Brazil's two leading universities: UNICAMP (Comvest) and USP (Fuvest), spanning exam years 2022--2025.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":50,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2606.22723"},"ranking":{"90d":{"score":15,"rank":407,"coverage":1.0,"confidence":"High","datasetDownloadRank":62,"datasetRankPopulation":66}},"description":"BLUEX v2 evaluates large language models on open-ended, discursive questions from the second-phase entrance exams of UNICAMP and USP (2022–2025). The dataset includes 395 questions with 919 subquestions, covering nine subjects, with 55.7% of questions containing images represented as context-aware captions. Scoring uses an LLM-as-a-judge protocol with binary rubric criteria based on official reference answers, yielding a 0–10 score.","whyItMatters":"Portuguese-language evaluation of LLMs has been limited, especially for open-ended tasks requiring deep reasoning and generation. This benchmark provides a public, reusable testbed for assessing capabilities in mathematical reasoning, image understanding, and other dimensions, offering comparable scores across models for practical model selection.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"25b9bea1583f3f0749a1aec78df6c8536a7f11a287be1b637f2ee899196b0067"},"motivation":"Although Large Language Models (LLMs) excel in many tasks, their assessment in Portuguese has received less attention, particularly for open-ended, discursive tasks that demand deeper reasoning and generation capabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22723","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TropicAI Research","organizationType":"academic-lab","sourceUrl":"https://github.com/TropicAI-Research/BLUEXv2","role":"benchmark-publisher"}],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_boiling-the-frog_8414ba51","familyId":"bmf_f1d48d3c8c74","name":"Boiling the Frog","oneLine":"Boiling the Frog evaluates whether tool-using AI models in corporate settings are susceptible to incremental attacks through multi-turn scenarios with persistent workspaces. It includes a three-level operational risk taxonomy and scores attack success rate (ASR) on resulting artifact state.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22643","pdf":"https://arxiv.org/pdf/2605.22643","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.22643"},"evidence":{"snippet":"We introduce Boiling the Frog, a benchmark that evaluates whether tool-using AI models deployed in corporate and office settings are susceptible to incremental attacks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.22643"},"ranking":{},"description":"Boiling the Frog evaluates whether tool-using AI models in corporate settings are susceptible to incremental attacks through multi-turn scenarios with persistent workspaces. It includes a three-level operational risk taxonomy and scores attack success rate (ASR) on resulting artifact state.","whyItMatters":"Addresses the gap in safety evaluation for agents acting in environments, focusing on incremental manipulation rather than single-turn textual outputs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fcdac064900bc9e92b7202a5484c1389a775db9ac286461cf3194aed80de665c"},"motivation":"Background.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.22643","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_df39c63037634966","familyId":"catalog_family_df39c63037634966","name":"BoolQ","oneLine":"BoolQ is a reading comprehension dataset for yes/no questions containing 15,942 naturally occurring examples. Each example consists of a question, passage, and boolean answer, where questions are generated in unprompted and unconstrained settings. The dataset challenges models with complex, non-factoid information requiring entailment-like inference to solve.","description":"BoolQ is a reading comprehension dataset for yes/no questions containing 15,942 naturally occurring examples. Each example consists of a question, passage, and boolean answer, where questions are generated in unprompted and unconstrained settings. The dataset challenges models with complex, non-factoid information requiring entailment-like inference to solve.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/boolq","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_df39c63037634966"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/boolq"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"boolq","url":"https://llm-stats.com/benchmarks/boolq","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning"],"catalogModelCount":10,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_boundary-bench_5415f6d9","familyId":"bmf_29b2f52f6f51","name":"Boundary-Bench","oneLine":"Boundary-Bench is an open-source plugin that adds configurable security policy levels to Terminal-Bench, enabling evaluation of coding agents under constraints like scoped credentials, restricted egress, read-only filesystems, and non-root execution.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CR"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02670","pdf":"https://arxiv.org/pdf/2608.02670","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.02670"},"evidence":{"snippet":"We release Boundary-Bench, an open-source hardening plugin enabling policy-constrained evaluation of coding agents on Terminal-Bench and compatible benchmarks.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02670"},"ranking":{"30d":{"score":41,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":45,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Boundary-Bench is an open-source plugin that adds configurable security policy levels to Terminal-Bench, enabling evaluation of coding agents under constraints like scoped credentials, restricted egress, read-only filesystems, and non-root execution.","whyItMatters":"Existing coding agent benchmarks assume permissive sandboxes, leaving a gap in understanding performance under real-world security policies. This benchmark provides a standardized way to measure success and efficiency trade-offs across policy levels, informing model selection for deployment in hardened environments.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9a2d68a81eda969254befd1930f2ad4b0a24d9426358d5cb66bb11f832e4ce48"},"motivation":"Coding agents increasingly run inside organizations whose security controls (scoped credentials, restricted egress, read-only filesystems, non-root execution) constrain them like any other software.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02670","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Boundary-Bench authors","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.02670","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_braillebench_3cc849b8","familyId":"bmf_b83912f7ae2f","name":"BrailleBench","oneLine":"Although Large language models (LLMs) mediate access to knowledge and computational assistance, their capabilities should benefit vulnerable groups in the same way.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.27268","pdf":"https://arxiv.org/pdf/2608.27268","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.27268"},"evidence":{"snippet":"To this end, we introduce BrailleBench, a benchmark for evaluating LLMs in Braille comprehension from different Criteria.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27268"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"Although Large language models (LLMs) mediate access to knowledge and computational assistance, their capabilities should benefit vulnerable groups in the same way.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.27268","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_brainbench_1fd81276","familyId":"bmf_4ec75908a077","name":"BrainBench","oneLine":"BrainBench is a benchmark for instruction-conditioned EEG understanding covering four subsets across 17 datasets. It evaluates LLMs on producing scientific reports and artifacts from EEG recordings.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04156","pdf":"https://arxiv.org/pdf/2608.04156","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.04156"},"evidence":{"snippet":"We introduce \\benchmarkname{}, a unified benchmark for comprehensive, instruction-conditioned EEG understanding.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04156"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"BrainBench is a benchmark for instruction-conditioned EEG understanding covering four subsets across 17 datasets. It evaluates LLMs on producing scientific reports and artifacts from EEG recordings.","whyItMatters":"Provides a unified evaluation for comprehensive EEG understanding, enabling comparison across models and execution paradigms, advancing LLM-based EEG analysis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ac297908643627604420be8cfe12469eecadbd7c1fe2abfc144b2a79e14b8442"},"motivation":"Electroencephalography (EEG) analysis extends beyond assigning predefined labels to recordings; it requires workflows connecting natural-language instructions, signal processing, quantitative evidence, and scientific interpretation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04156","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"lib_browsecomp","familyId":"family_browsecomp","name":"BrowseComp","oneLine":"Established benchmark family · Agents & Tool Use.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents & Tool Use"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2025-01-01","releaseDatePrecision":"year","firstRelease":{"year":2025,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2504.12516","pdf":null,"project":"https://openai.com/index/browsecomp/","code":"https://github.com/openai/simple-evals","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_browsecomp"},"ranking":{},"recordType":"family","aliases":[],"sourceAttribution":[{"role":"official-release","url":"https://openai.com/index/browsecomp/"}],"adoptionRefs":["openai-gpt5"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"browseComp","url":"https://benchlm.ai/benchmarks/browsecomp","paperUrl":"https://openai.com/index/browsecomp/","year":"2025","fullName":"BrowseComp","format":"Web search and evidence synthesis","tasks":"Research questions requiring browsing","successorKey":null},{"catalog":"llm-stats","sourceId":"browsecomp","url":"https://llm-stats.com/benchmarks/browsecomp","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","search","agents"],"catalogModelCount":62,"catalogStarCount":0},{"id":"catalog_c30f789f0076a90d","familyId":"catalog_family_c30f789f0076a90d","name":"BrowseComp (10-agent, prerelease)","oneLine":"BrowseComp accuracy from ten collaborating Opus 5 agents on a pre-release model and unreleased effort configuration.","description":"BrowseComp accuracy from ten collaborating Opus 5 agents on a pre-release model and unreleased effort configuration.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c30f789f0076a90d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/multiagentbrowsecompprerelease"}],"catalogSources":[{"catalog":"benchlm","sourceId":"multiAgentBrowseCompPrerelease","url":"https://benchlm.ai/benchmarks/multiagentbrowsecompprerelease","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Multi-Agent BrowseComp — 10-agent team prerelease configuration","format":"10-agent team accuracy","tasks":"BrowseComp web-research tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_01284e3412ceb8cb","familyId":"catalog_family_01284e3412ceb8cb","name":"BrowseComp Long Context 128k","oneLine":"A challenging benchmark for evaluating web browsing agents' ability to persistently navigate the internet and find hard-to-locate, entangled information. Comprises 1,266 questions requiring strategic reasoning, creative search, and interpretation of retrieved content, with short and easily verifiable answers.","description":"A challenging benchmark for evaluating web browsing agents' ability to persistently navigate the internet and find hard-to-locate, entangled information. Comprises 1,266 questions requiring strategic reasoning, creative search, and interpretation of retrieved content, with short and easily verifiable answers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Search"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/browsecomp-long-128k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_01284e3412ceb8cb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/browsecomp-long-128k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"browsecomp-long-128k","url":"https://llm-stats.com/benchmarks/browsecomp-long-128k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","search"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_624989cc95dc50bc","familyId":"catalog_family_624989cc95dc50bc","name":"BrowseComp Long Context 256k","oneLine":"BrowseComp is a benchmark for measuring the ability of agents to browse the web, comprising 1,266 questions that require persistently navigating the internet in search of hard-to-find, entangled information. Despite the difficulty of the questions, BrowseComp is simple and easy-to-use, as predicted answers are short and easily verifiable against reference answers. The benchmark focuses on questions where answers are obscure, time-invariant, and well-supported by evidence scattered across the open web.","description":"BrowseComp is a benchmark for measuring the ability of agents to browse the web, comprising 1,266 questions that require persistently navigating the internet in search of hard-to-find, entangled information. Despite the difficulty of the questions, BrowseComp is simple and easy-to-use, as predicted answers are short and easily verifiable against reference answers. The benchmark focuses on questions where answers are obscure, time-invariant, and well-supported by evidence scattered across the open web.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Search"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/browsecomp-long-256k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_624989cc95dc50bc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/browsecomp-long-256k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"browsecomp-long-256k","url":"https://llm-stats.com/benchmarks/browsecomp-long-256k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","search"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_c6c0e2526edde416","familyId":"catalog_family_c6c0e2526edde416","name":"BrowseComp-VL","oneLine":"BrowseComp-VL is the vision-language variant of BrowseComp, evaluating multimodal models on web browsing comprehension tasks that require processing visual web page content alongside text.","description":"BrowseComp-VL is the vision-language variant of BrowseComp, evaluating multimodal models on web browsing comprehension tasks that require processing visual web page content alongside text.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Multimodal","Search","Agents","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c6c0e2526edde416"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/browsecompvl"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/browsecomp-vl"}],"catalogSources":[{"catalog":"benchlm","sourceId":"browseCompVl","url":"https://benchlm.ai/benchmarks/browsecompvl","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"BrowseComp-VL","format":"Vision-language web research evaluation","tasks":"Multimodal browsing tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"browsecomp-vl","url":"https://llm-stats.com/benchmarks/browsecomp-vl","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","multimodal","search","agents","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_4cbf3f4140d31be6","familyId":"catalog_family_4cbf3f4140d31be6","name":"BrowseComp-zh","oneLine":"A high-difficulty benchmark purpose-built to comprehensively evaluate LLM agents on the Chinese web, consisting of 289 multi-hop questions spanning 11 diverse domains including Film & TV, Technology, Medicine, and History. Questions are reverse-engineered from short, objective, and easily verifiable answers, requiring sophisticated reasoning and information reconciliation beyond basic retrieval. The benchmark addresses linguistic, infrastructural, and censorship-related complexities in Chinese web environments.","description":"A high-difficulty benchmark purpose-built to comprehensively evaluate LLM agents on the Chinese web, consisting of 289 multi-hop questions spanning 11 diverse domains including Film & TV, Technology, Medicine, and History. Questions are reverse-engineered from short, objective, and easily verifiable answers, requiring sophisticated reasoning and information reconciliation beyond basic retrieval. The benchmark addresses linguistic, infrastructural, and censorship-related complexities in Chinese web environments.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Search"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/browsecomp-zh","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4cbf3f4140d31be6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/browsecomp-zh"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"browsecomp-zh","url":"https://llm-stats.com/benchmarks/browsecomp-zh","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","search"],"catalogModelCount":13,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_40e218fcbdb18b98","familyId":"catalog_family_40e218fcbdb18b98","name":"BRUMO 2025","oneLine":"A challenging mathematical olympiad competition featuring problems that test advanced mathematical reasoning and problem-solving skills at the olympiad level.","description":"A challenging mathematical olympiad competition featuring problems that test advanced mathematical reasoning and problem-solving skills at the olympiad level.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.math.bas.bg/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_40e218fcbdb18b98"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/brumo2025"}],"catalogSources":[{"catalog":"benchlm","sourceId":"brumo2025","url":"https://benchlm.ai/benchmarks/brumo2025","paperUrl":"https://www.math.bas.bg/","year":"2025","fullName":"Bulgarian Mathematical Olympiad 2025","format":"Mathematical olympiad","tasks":"Olympiad problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_bts-agentbench_70852571","familyId":"bmf_9fae19317f65","name":"BTS-AgentBench","oneLine":"Evaluates multi-turn agent performance on 532 telemetry-derived tasks across train/dev/test splits, with additional XAI4HEAT episodes, using deterministic replayable construction and verifiable gold answers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":0.75,"links":{"report":"https://arxiv.org/abs/2608.27334","pdf":"https://arxiv.org/pdf/2608.27334","project":null,"code":"https://github.com/kjy7567/BTS-AgentBench","data":null,"hfPaper":null},"evidence":{"snippet":"Industrial sites contain large volumes of read-only telemetry, but few benchmarks specify how to compile these records into executable multi-turn agent tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":null,"hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27334"},"ranking":{},"description":"Evaluates multi-turn agent performance on 532 telemetry-derived tasks across train/dev/test splits, with additional XAI4HEAT episodes, using deterministic replayable construction and verifiable gold answers.","whyItMatters":"Provides a reproducible pipeline from raw telemetry to agent benchmarks, enabling consistent evaluation of agent capabilities in industrial settings with evidence attribution and quality-gated reporting.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T07:23:36.749451Z","inputHash":"df21106de8807ca0668c46fdb6499344a0f540394e336611d39a6f15c3d46f0d"},"motivation":"Industrial sites contain large volumes of read-only telemetry, but few benchmarks specify how to compile these records into executable multi-turn agent tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T07:23:36.749451Z","model":"deepseek-v4-pro","decisionReason":"The paper releases code, artifacts, and replay reports at a public GitHub repository, with a stable scoring contract, public reuse path, and explicit benchmark artifacts.","canonicalNameSource":"paper_title","canonicalNameEvidence":"BTS-AgentBench: A Deterministic, Replayable Pipeline from Read-Only Telemetry Logs to Agent Benchmarks"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.27334","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-30T06:59:55.242506Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses industrial agent evaluation with a reproducible pipeline, likely to attract attention from researchers and practitioners in agent benchmarking."},"evaluationMode":"public_reusable","publishers":[{"name":"BTS-AgentBench authors","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.27334","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_bugsourcebench_7e234238","familyId":"bmf_7f56fcdc5c5d","name":"BugSourceBench","oneLine":"The work introduces BugSourceBench, a code repair benchmark with bugs from human-written, LM-generated, and human-edited LM-generated code. The benchmark evaluates the fix rate of language models on these bug sources. No scoring contract, dataset, or public access path is provided.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.03523","pdf":"https://arxiv.org/pdf/2607.03523","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.03523"},"evidence":{"snippet":"To test whether this curriculum generalizes, we introduce BugSourceBench, a repair benchmark spanning realistic bug sources: bugs in human-written code, LM-generated code, and human-edited LM-generated code.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.03523"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"The work introduces BugSourceBench, a code repair benchmark with bugs from human-written, LM-generated, and human-edited LM-generated code. The benchmark evaluates the fix rate of language models on these bug sources. No scoring contract, dataset, or public access path is provided.","whyItMatters":"Code repair benchmarks often focus on synthetic bugs, which may not reflect real-world failures. BugSourceBench aims to cover diverse bug origins, potentially offering a more realistic evaluation for repair models. However, the lack of accessible artifacts and evaluation protocol limits its current utility.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b9c19a3411cf118e7b6ffe7f130362be00567913ec033ad6be4d7c635e51121f"},"motivation":"Code repair is an important capability for language models (LMs): given a buggy program and unit tests, an LM must produce a fixed program that passes the tests.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.03523","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_bulkpr-bench_1253579f","familyId":"bmf_10e04fc9ee92","name":"BulkPR-Bench","oneLine":"BulkPR-Bench evaluates governance of interacting pull requests. It includes 581 candidate PRs on 18 repositories, with metrics RDS and Global-SGY measuring safe delivery. The benchmark uses executable repository execution with hidden safety checks.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02685","pdf":"https://arxiv.org/pdf/2608.02685","project":"https://doi.org/10.5281/zenodo.21717780","code":"https://github.com/Eureka246/BulkPR-Bench-Release","data":null,"hfPaper":"https://huggingface.co/papers/2608.02685"},"evidence":{"snippet":"We introduce BulkPR-Bench, an executable benchmark in which an agent must recover consequential PR relations and return a large safe subset in executable order under a rolling-release protocol.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02685"},"ranking":{"30d":{"score":23,"rank":158,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":362,"coverage":0.55,"confidence":"Low"}},"description":"BulkPR-Bench evaluates governance of interacting pull requests. It includes 581 candidate PRs on 18 repositories, with metrics RDS and Global-SGY measuring safe delivery. The benchmark uses executable repository execution with hidden safety checks.","whyItMatters":"Coding-agent benchmarks often assume independent PRs. BulkPR-Bench addresses interactive PR queues, providing a benchmark for evaluating joint decision-making in complex scenarios, with executable validation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"12c3924e5aee28a796c6e54f3bd0c7d13732d61f8f27783829c5fdb4b37cccb6"},"motivation":"Coding-agent benchmarks increasingly cover long-horizon, end-to-end, and interactive development, but typically retain one requested outcome or a fixed change sequence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02685","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Zenodo","organizationType":"benchmark-organization","sourceUrl":"https://doi.org/10.5281/zenodo.21717780","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_149a7a2b6d8a6bc9","familyId":"catalog_family_149a7a2b6d8a6bc9","name":"BullshitBench v2","oneLine":"A benchmark that tests whether AI models challenge nonsensical, ill-posed, or logically flawed prompts instead of confidently generating incorrect answers. Measures the critical ability to push back on bad input.","description":"A benchmark that tests whether AI models challenge nonsensical, ill-posed, or logically flawed prompts instead of confidently generating incorrect answers. Measures the critical ability to push back on bad input.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://petergpt.github.io/bullshit-benchmark/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_149a7a2b6d8a6bc9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/bullshitbenchv2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"bullshitBenchV2","url":"https://benchlm.ai/benchmarks/bullshitbenchv2","paperUrl":"https://petergpt.github.io/bullshit-benchmark/","year":"2025","fullName":"BullshitBench v2","format":"Prompt challenge and refusal evaluation","tasks":"Nonsensical and flawed prompts across multiple domains","successorKey":null}],"catalogCategories":["reasoning"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e32e7c8c89936ac7","familyId":"catalog_family_e32e7c8c89936ac7","name":"C-Eval","oneLine":"C-Eval is a comprehensive Chinese evaluation suite designed to assess advanced knowledge and reasoning abilities of foundation models in a Chinese context. It comprises 13,948 multiple-choice questions across 52 diverse disciplines spanning humanities, science, and engineering, with four difficulty levels: middle school, high school, college, and professional. The benchmark includes C-Eval Hard, a subset of very challenging subjects requiring advanced reasoning abilities.","description":"C-Eval is a comprehensive Chinese evaluation suite designed to assess advanced knowledge and reasoning abilities of foundation models in a Chinese context. It comprises 13,948 multiple-choice questions across 52 diverse disciplines spanning humanities, science, and engineering, with four difficulty levels: middle school, high school, college, and professional. The benchmark includes C-Eval Hard, a subset of very challenging subjects requiring advanced reasoning abilities.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2305.08322","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e32e7c8c89936ac7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/ceval"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/c-eval"}],"catalogSources":[{"catalog":"benchlm","sourceId":"cEval","url":"https://benchlm.ai/benchmarks/ceval","paperUrl":"https://arxiv.org/abs/2305.08322","year":"2023","fullName":"C-Eval","format":"Multiple choice questions","tasks":"Chinese academic and professional exams","successorKey":null},{"catalog":"llm-stats","sourceId":"c-eval","url":"https://llm-stats.com/benchmarks/c-eval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","general"],"catalogModelCount":18,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_c-suitebench_9e6fa218","familyId":"bmf_77b58031dbe0","name":"C-SUITEBENCH","oneLine":"C-SUITEBENCH evaluates multimodal LLMs as CEOs on five decision tasks under paired text-only and multimodal conditions across 50 scenarios, focusing on evidence-centric reasoning and constraint satisfaction.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.05864","pdf":"https://arxiv.org/pdf/2608.05864","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.05864"},"evidence":{"snippet":"We introduce C-SUITEBENCH, a controlled multimodal benchmark that includes five decision tasks under paired text-only and multimodal conditions across 50 scenarios.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.05864"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"C-SUITEBENCH evaluates multimodal LLMs as CEOs on five decision tasks under paired text-only and multimodal conditions across 50 scenarios, focusing on evidence-centric reasoning and constraint satisfaction.","whyItMatters":"Existing executive decision benchmarks are text-only, leaving unclear whether models can integrate visual evidence. C-SUITEBENCH reveals a multimodal integration paradox where visual inputs can degrade constrained allocation, informing selective grounding strategies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"042d1bc558a788a5b6c088e56b9e4d4379e49eda315bc451067f00da7927e8d9"},"motivation":"Large language models are increasingly applied as autonomous decision-making agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.05864","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_c3-bench_c82d4fcd","familyId":"bmf_8434d033ab49","name":"C3-Bench","oneLine":"C3-Bench evaluates context-aware change captioning with 4,996 human-labeled image pairs across 51 real-world contexts, using an LLM-as-Judge framework scoring correctness, specificity, fluency, relevance, and a reversibility metric.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.25445","pdf":"https://arxiv.org/pdf/2606.25445","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.25445"},"evidence":{"snippet":"To fill this gap, we propose C3-Bench, a comprehensive benchmark for evaluating Context-aware Change Captioning.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.25445"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"C3-Bench evaluates context-aware change captioning with 4,996 human-labeled image pairs across 51 real-world contexts, using an LLM-as-Judge framework scoring correctness, specificity, fluency, relevance, and a reversibility metric.","whyItMatters":"Change captioning performance varies with domain; C3-Bench exposes systematic failures in conventional models and LMMs, providing a comprehensive benchmark to drive generalization and trustworthiness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"30ec2ca6804caee6ea6be21a47bec4b8e687d6de1e05519e737c92956db04115"},"motivation":"While Change Captioning systems have garnered substantial attention to respond to our evolving world, their true performance on diverse real-world change contexts remains largely unexplored due to the lack of comprehensive evaluation frameworks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.25445","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_cadengbench_ba1f84a8","familyId":"bmf_3e931326d51e","name":"CADEngBench","oneLine":"CADEngBench evaluates parametric CAD part generation and editing through B-Rep validity, engineering checks, parameter perturbation, and linear-static FEA, plus assembly reasoning via joint retrieval, grounding, and kinematic verification.","area":"Science & Engineering","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Geometric reasoning"],"topics":["CAD","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09296","pdf":"https://arxiv.org/pdf/2608.09296","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09296"},"evidence":{"snippet":"We present CADEngBench, a two-track benchmark for these capabilities.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09296"},"ranking":{"30d":{"score":39,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"CADEngBench evaluates parametric CAD part generation and editing through B-Rep validity, engineering checks, parameter perturbation, and linear-static FEA, plus assembly reasoning via joint retrieval, grounding, and kinematic verification.","whyItMatters":"It addresses the gap in CAD evaluation by focusing on engineering behavior rather than visual appearance, providing a structured test for design validity, functional editing, and assembly correctness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c2109b05e810c3ab40d8f744de90c7a2e1b2c291d4fed0a2371855efb762ad1e"},"motivation":"A CAD model is not engineering-grade merely because it looks correct.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09296","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_cadp-bench_c4c34058","familyId":"bmf_a82a8535ed6e","name":"CADP-Bench","oneLine":"Evaluates MLLMs that reconstruct full academic pages as compilable LaTeX and executable Python from page images.","area":"Multimodal","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":["Academic document parsing","Structured document reconstruction","Chart-to-code generation","Multimodal code generation"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.17550","pdf":"https://arxiv.org/pdf/2608.17550","project":"https://github.com/AriKing11/CADP-Bench","code":"https://github.com/AriKing11/CADP-Bench","data":"https://github.com/AriKing11/CADP-Bench/tree/main/data","hfPaper":"https://huggingface.co/papers/2608.17550"},"evidence":{"snippet":"To support this setting, we introduce CADP-Bench, an expert-verified benchmark of full academic pages containing tightly coupled text and multiple SAE types, evaluated through a re-injection compilation protocol.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17550"},"ranking":{"30d":{"score":28,"rank":93,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":263,"coverage":0.55,"confidence":"Low"}},"description":"CADP-Bench evaluates multimodal LLMs on compilable academic document parsing. It includes expert-verified full academic pages with tightly coupled text and structured elements, assessed through a re-injection compilation protocol.","whyItMatters":"Provides a structured evaluation for structure-aware scientific document parsing, measuring the fidelity of executable reconstructions, which is crucial for machine-readable scientific knowledge.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"85b405ab2d375c2799108825ab66f088802f9ba6930377405a31d41e0d220d9e"},"motivation":"Academic papers are a primary carrier of scientific knowledge, yet most of this knowledge remains locked in PDFs that are optimized for human reading rather than machine use.","constructionDetail":"CADP-Bench asks models to reconstruct complete academic pages as compilable LaTeX and executable Python, then scores both structure and rendered output.","detail":{"taskBreakdown":["Computer Science","Physics","Economics","Quantitative Biology","Statistics"],"protocol":{"tasks":"1,630 full-page samples","primaryMetric":"Pixel Similarity with structural, reading-order and execution metrics","version":"v1"}},"curation":{"state":"source-reviewed","reviewedAt":"2026-08-20","sources":["https://arxiv.org/abs/2608.17550","https://arxiv.org/html/2608.17550v1","https://github.com/AriKing11/CADP-Bench"]},"publication":{"status":"acceptance_claimed","venue":"ACM MM 2026","evidence":"Accepted by ACM MM 2026","evidenceUrl":"https://arxiv.org/abs/2608.17550","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ACM MM 2026","reviewStatus":"accepted","decisionRaw":"Accepted by ACM MM 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.17550","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by ACM MM 2026","level":"author-claim"}]}],"evaluationMode":"score_submission","availability":{"paperStatus":"available","githubStatus":"available","hfDatasetStatus":"sample_only","evaluatorStatus":"available","submissionStatus":"not_found"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_cafosat_e236d412","familyId":"bmf_01528fdfb43e","name":"CAFOSat","oneLine":"CAFOSat is a dataset of over 45,000 image patches with infrastructure-level annotations for CAFO mapping across 20 states, benchmarking models for infrastructure-aware agricultural monitoring.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.00548","pdf":"https://arxiv.org/pdf/2606.00548","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.00548"},"evidence":{"snippet":"We benchmark a diverse set of convolutional, transformer-based, and vision-language models, demonstrating the value of refined annotations and curated negative samples for CAFO classification and generalization.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00548"},"ranking":{},"description":"CAFOSat is a dataset of over 45,000 image patches with infrastructure-level annotations for CAFO mapping across 20 states, benchmarking models for infrastructure-aware agricultural monitoring.","whyItMatters":"Provides a large-scale, infrastructure-aware benchmark for advancing CAFO mapping from remote sensing, addressing the lack of strongly annotated datasets in this domain.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"50c6afb0efd9947517d58fa6465084e6d3546ec1ae8d1fcf9532b1f60ff2aadc"},"motivation":"Concentrated Animal Feeding Operations (CAFOs) play an important role in agricultural production but are also associated with environmental, public health, and disease surveillance concerns.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"CVPR Workshop-2026","evidence":"Accepted at CVPR Workshop-2026. First two authors has equal contribution","evidenceUrl":"https://arxiv.org/abs/2606.00548","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"CVPR Workshop-2026","reviewStatus":"accepted","decisionRaw":"Accepted at CVPR Workshop-2026. First two authors has equal contribution","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.00548","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at CVPR Workshop-2026. First two authors has equal contribution","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_dfcaa4c5de48a149","familyId":"catalog_family_dfcaa4c5de48a149","name":"CAIS Text Leaderboard","oneLine":"A Center for AI Safety dashboard view summarizing text capabilities across HLE, ARC-AGI-2, SWE-Bench Pro, and TextQuests.","description":"A Center for AI Safety dashboard view summarizing text capabilities across HLE, ARC-AGI-2, SWE-Bench Pro, and TextQuests.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://dashboard.safe.ai/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dfcaa4c5de48a149"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/caistextleaderboard"}],"catalogSources":[{"catalog":"benchlm","sourceId":"caisTextLeaderboard","url":"https://benchlm.ai/benchmarks/caistextleaderboard","paperUrl":"https://dashboard.safe.ai/","year":"2025","fullName":"CAIS AI Dashboard Text Capabilities Index","format":"Average component score","tasks":"HLE, ARC-AGI-2, SWE-Bench Pro, and TextQuests","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_callbench_09a4fc23","familyId":"bmf_502e5ff2677a","name":"CallBench","oneLine":"CallBench is a Chinese benchmark for evaluating dual-goal coordination in phone call assistants, with 50,000 multi-turn dialogues across six scenarios and a preset-aware turn-level scoring protocol.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.22635","pdf":"https://arxiv.org/pdf/2607.22635","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.22635"},"evidence":{"snippet":"We introduce \\textsc{CallBench}, a Chinese benchmark for evaluating dual-goal coordination in phone call assistants.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.22635"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CallBench is a Chinese benchmark for evaluating dual-goal coordination in phone call assistants, with 50,000 multi-turn dialogues across six scenarios and a preset-aware turn-level scoring protocol.","whyItMatters":"Existing dialogue benchmarks focus on single explicit goals, while real phone assistants must balance the owner's preset and the caller's dynamic goal. CallBench provides a reusable evaluation to measure turn-level decisions under proxy constraints.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3825108a9b4ac0a2ddb8111fde25627ed9cd0a9c8ea24fa29d5a269e26dbf0f7"},"motivation":"Target-oriented dialogue systems have demonstrated strong capabilities in completing user goals through interactive conversations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.22635","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_caloriebench-80k_97ce8906","familyId":"bmf_8745ff739a30","name":"CalorieBench-80K","oneLine":"CalorieBench-80K is a food image benchmark with calorie labels and dietary advice annotations, used to evaluate Food-R1 and other models for food analysis tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04986","pdf":"https://arxiv.org/pdf/2606.04986","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04986"},"evidence":{"snippet":"To address these issues, we introduce CalorieBench-80K, a large-scale benchmark with curated calorie labels and dietary advice annotations.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04986"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CalorieBench-80K is a food image benchmark with calorie labels and dietary advice annotations, used to evaluate Food-R1 and other models for food analysis tasks.","whyItMatters":"Provides a large-scale food benchmark with chain-of-thought annotations for calorie reasoning, enabling evaluation of VLMs on nutritional understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f6d88539d530530e0a18888868cffce5736f74f94888547c55371236d002ab62"},"motivation":"Recent studies have explored Vision-Language Models (VLMs) for food analysis.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04986","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_temporally-grounded-compositional-camera-m_d9aa8053","familyId":"bmf_57f7e4349ad6","name":"CamChoreo","oneLine":"Evaluates temporally grounded compositional camera motion understanding using 4,229 single-shot clips with expert-annotated segments and 20 direction-aware labels.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.10932","pdf":"https://arxiv.org/pdf/2608.10932","project":"https://ddz16.github.io/cammotion.github.io/","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce CamChoreo, a benchmark of 4,229 real single-shot clips with expert-annotated temporal segments.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10932"},"ranking":{"30d":{"score":19,"rank":162,"coverage":0.85,"confidence":"High"},"90d":{"score":25,"rank":295,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates temporally grounded compositional camera motion understanding using 4,229 single-shot clips with expert-annotated segments and 20 direction-aware labels.","whyItMatters":"Advances camera motion evaluation beyond clip-level labels to compositional and temporal localization, exposing limitations in current multimodal models.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"85160b322deec9893e14396466a4c728662b7194e488688841f96bbf38147ff0"},"motivation":"Understanding camera motion is fundamental to video perception, with applications in spatial intelligence and controllable video generation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"The paper defines a formal benchmark with expert annotations and a project page, providing a reusable dataset and evaluation task.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce CamChoreo, a benchmark of 4,229 real single-shot clips with expert-annotated temporal segments"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10932","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":52,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses multimodal video understanding with a clear public project page and detailed methodology, likely to draw moderate attention from researchers in video and MLLM evaluation."},"evaluationMode":"public_reusable","publishers":[{"name":"CamChoreo Benchmark Team","organizationType":"academic-lab","sourceUrl":"https://ddz16.github.io/cammotion.github.io/","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_canlegalragbench_9244f079","familyId":"bmf_437f1c13e33f","name":"CanLegalRAGBench","oneLine":"CanLegalRAGBench evaluates retrieval-augmented generation for Canadian case law with realistic queries and expert-annotated answers. It tests retrieval and answer generation grounded in legal documents.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.30497","pdf":"https://arxiv.org/pdf/2605.30497","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30497"},"evidence":{"snippet":"To address this gap, we introduce CanLegalRAGBench, a Canadian legal QA benchmark based on realistic queries and expert-annotated answers grounded in case law.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30497"},"ranking":{},"description":"CanLegalRAGBench evaluates retrieval-augmented generation for Canadian case law with realistic queries and expert-annotated answers. It tests retrieval and answer generation grounded in legal documents.","whyItMatters":"Fills a gap in legal RAG evaluation for Canadian law, providing realistic scenarios and revealing limitations in automatic metrics and hallucination in generated answers.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"aaa1591f1e915737f9bc08528854dc7d2affab1dc96ae75f49c9cc76210c12ec"},"motivation":"RAG-based legal assistants have been growing in popularity, but LLM hallucinations remain a key issue and potentially undermines justice.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30497","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_cann-bench_f89ecd9f","familyId":"bmf_521039f15a2e","name":"CANN Bench","oneLine":"Proposes a benchmark for AI-generated operator code on Huawei's Ascend NPU, covering 53 operators and 1060 test cases. Evaluation uses a three-dimensional weighted composite score for compilation, functional correctness, and performance.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20518","pdf":"https://arxiv.org/pdf/2607.20518","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20518"},"evidence":{"snippet":"We present CANN Bench, an open benchmark for AI-generated operator code on Huawei's Ascend NPU.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20518"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Proposes a benchmark for AI-generated operator code on Huawei's Ascend NPU, covering 53 operators and 1060 test cases. Evaluation uses a three-dimensional weighted composite score for compilation, functional correctness, and performance.","whyItMatters":"Could fill a gap in evaluating kernel-generation agents beyond CUDA/Triton, but clear artifact availability and a usable scoring contract are not confirmed from provided evidence.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"78b1e5069750a99098f386f8e76081436b294d5dbfa338943023ca32e02512a4"},"motivation":"AI agents are now capable of writing, compiling, and iteratively optimizing low-level operator kernels on different hardware platforms.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20518","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_cap_c96b1f01","familyId":"bmf_45e84635bc48","name":"CAP","oneLine":"CAP evaluates cross-site browser agents on 420 tasks across 108 real websites and 24 domains. Tasks require complex UI interactions and visual perception, with scoring via an agent-as-a-judge framework.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.08392","pdf":"https://arxiv.org/pdf/2608.08392","project":"https://warriorxu0302.github.io/CAP-Bench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.08392"},"evidence":{"snippet":"We introduce CAP, a scalable benchmark for evaluating browser agents on cross-site, human-like web tasks that require non-trivial UI interactions and visual understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08392"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CAP evaluates cross-site browser agents on 420 tasks across 108 real websites and 24 domains. Tasks require complex UI interactions and visual perception, with scoring via an agent-as-a-judge framework.","whyItMatters":"CAP addresses the gap in browser agent evaluation by focusing on cross-site workflows and perception-heavy interactions, which are common in real-world browsing. It provides fine-grained diagnostics to identify bottlenecks in current agents, aiding targeted improvements.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2477e8d117b2c7ccb623acd131c8c26a2ed1022dbd26a5b62590a12880799d86"},"motivation":"Large language models are increasingly deployed as autonomous agents that interact with the web through browsers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"COLM 2026","evidence":"Accepted to COLM 2026. Project page: https://warriorxu0302.github.io/CAP-Bench/","evidenceUrl":"https://arxiv.org/abs/2608.08392","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"COLM 2026","reviewStatus":"accepted","decisionRaw":"Accepted to COLM 2026. Project page: https://warriorxu0302.github.io/CAP-Bench/","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.08392","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to COLM 2026. Project page: https://warriorxu0302.github.io/CAP-Bench/","level":"author-claim"}]}],"publishers":[{"name":"CAP-Bench Project","organizationType":"academic-lab","sourceUrl":"https://warriorxu0302.github.io/CAP-Bench/","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_capprobe_99438ec2","familyId":"bmf_1dc0b6750669","name":"CapProbe","oneLine":"CapProbe is a full-scene dense QA benchmark for evaluating detailed image captions from Vision-Language Models. It decomposes images into semantic regions and generates multiple-choice questions across 10 semantic categories, with a language judge answering from captions. The benchmark comprises 346 images, 1,868 regions, and 25,650 questions.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.11074","pdf":"https://arxiv.org/pdf/2608.11074","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.11074"},"evidence":{"snippet":"We introduce CapProbe, a full-scene dense QA benchmark that turns detailed caption evaluation into region-aligned factual checking.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11074"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CapProbe is a full-scene dense QA benchmark for evaluating detailed image captions from Vision-Language Models. It decomposes images into semantic regions and generates multiple-choice questions across 10 semantic categories, with a language judge answering from captions. The benchmark comprises 346 images, 1,868 regions, and 25,650 questions.","whyItMatters":"Existing metrics for detailed caption evaluation struggle to verify dense factual claims. CapProbe addresses this by region-aligned factual checking with dense QA, offering a cost-effective protocol that reduces open-ended scoring bias and reveals coverage gaps and trade-offs across models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0b0ea05a9983cd82fa70f568d7715455979e4e4beca8085fd1dea1bd6387fc1f"},"motivation":"Evaluating detailed image captions from Vision-Language Models (VLMs) requires going beyond surface-level semantic similarity.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11074","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_capricorn-1k_86185f2f","familyId":"bmf_ea863771b451","name":"CapRiCorn-1K","oneLine":"CapRiCorn-1K evaluates video captioning quality and subject referential consistency across long videos (15s-10min) with audiovisual and visual-only settings. It uses LLM judge to compute accuracy, coverage, and referential consistency metrics based on manual annotations.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.21949","pdf":"https://arxiv.org/pdf/2606.21949","project":null,"code":"https://github.com/xlchen0205/CapRiCorn-1K","data":null,"hfPaper":"https://huggingface.co/papers/2606.21949"},"evidence":{"snippet":"To bridge this gap, we propose CapRiCorn-1K, a comprehensive benchmark designed to evaluate both video captioning quality and subject referential consistency across long temporal horizons and diverse video domains.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.21949"},"ranking":{"90d":{"score":15,"rank":404,"coverage":0.7,"confidence":"Medium"}},"description":"CapRiCorn-1K evaluates video captioning quality and subject referential consistency across long videos (15s-10min) with audiovisual and visual-only settings. It uses LLM judge to compute accuracy, coverage, and referential consistency metrics based on manual annotations.","whyItMatters":"Existing benchmarks focus on short videos and overall caption quality, lacking evaluation of subject referential consistency over long horizons. CapRiCorn-1K's metrics correlate with downstream understanding and generation performance, offering practical value for selecting captioning models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"41a9d510b9fcfdd080442ff8f92514749267c03bfdd2dedd97a3e59949083339"},"motivation":"Accurate and comprehensive video captions with consistent subject references are critical for downstream understanding and generation tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.21949","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"CapRiCorn-1K Team","organizationType":"academic-lab","sourceUrl":"https://github.com/xlchen0205/CapRiCorn-1K","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_9d542e1869409972","familyId":"catalog_family_9d542e1869409972","name":"Capture-the-Flag Challenges (Internal)","oneLine":"Capture-the-Flag Challenges is OpenAI's internal expansion of competitive, professional-level cybersecurity capture-the-flag tasks used to evaluate vulnerability identification and exploitation capability under the Preparedness Framework.","description":"Capture-the-Flag Challenges is OpenAI's internal expansion of competitive, professional-level cybersecurity capture-the-flag tasks used to evaluate vulnerability identification and exploitation capability under the Preparedness Framework.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/capture-the-flag-challenges","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9d542e1869409972"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/capture-the-flag-challenges"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"capture-the-flag-challenges","url":"https://llm-stats.com/benchmarks/capture-the-flag-challenges","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety","agents","code"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_captureguide-bench_fb3878c6","familyId":"bmf_d80d29d0b93a","name":"CaptureGuide-Bench","oneLine":"CaptureGuide-Bench evaluates capture-time photography guidance with two tasks: photographer-side composition decision/refinement and subject-side scene-conditioned pose recommendation, using metrics like IoU and plausibility.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.25763","pdf":"https://arxiv.org/pdf/2606.25763","project":"https://lijayutnt.github.io/ShutterMuse","code":"https://github.com/lijayuTnT/ShutterMuse","data":null,"hfPaper":"https://huggingface.co/papers/2606.25763"},"evidence":{"snippet":"To address this gap, we introduce CaptureGuide-Bench, a benchmark with two complementary tasks: photographer-side composition decision and refinement, and subject-side scene-conditioned pose recommendation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":46,"hfDailySubmittedAt":"2026-06-25T00:00:00.000Z","githubStars":100,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.25763"},"ranking":{"90d":{"score":60,"rank":19,"coverage":0.7,"confidence":"Medium"}},"description":"CaptureGuide-Bench evaluates capture-time photography guidance with two tasks: photographer-side composition decision/refinement and subject-side scene-conditioned pose recommendation, using metrics like IoU and plausibility.","whyItMatters":"Existing aesthetic benchmarks focus on post-hoc cropping; CaptureGuide-Bench addresses missing capture-time guidance for both framing and subject pose, providing a structured dataset for development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4460e590ef6bd3ff14cfd2afd21bb19ed0562be30bb15c8446d1711ff37a277d"},"motivation":"Real-world photography requires capture-time guidance for both camera framing and subject pose.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.25763","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_cardiolens_7981754b","familyId":"bmf_f77ee8f96dfc","name":"CardioLens","oneLine":"CardioLens evaluates MLLMs on multi-sequence cardiac MRI interpretation, covering image understanding, report generation, and disease diagnosis using QA pairs from private hospital archives.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00123","pdf":"https://arxiv.org/pdf/2606.00123","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.00123"},"evidence":{"snippet":"We introduce CardioLens, a leakage-resistant evaluation testbed for multi-sequence Cardiovascular Magnetic Resonance (CMR), constructed from private hospital archives through a rigorous report-to-QA construction and verification pipeline.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00123"},"ranking":{},"description":"CardioLens evaluates MLLMs on multi-sequence cardiac MRI interpretation, covering image understanding, report generation, and disease diagnosis using QA pairs from private hospital archives.","whyItMatters":"The benchmark highlights the gap between public medical benchmark performance and clinical use, but its private data and lack of a public reuse path limit its value for broader model comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5524ca96b88b4e8a8b79ccf8ed616518e4163d04fde6c199850165536658c0bd"},"motivation":"Multimodal Large Language Models (MLLMs) have shown strong performance on public medical benchmarks, yet existing evaluations often remain weak proxies for clinical use, relying on isolated inputs and simplified recognition-style tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00123","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_care-bench_02aa6f70","familyId":"bmf_1e30cd79e194","name":"CARE-Bench","oneLine":"CARE-Bench is a source-grounded benchmark for patient-facing triage evaluation with 500 cases and 1,059 patient-disclosure prefixes. It evaluates LLMs on a four-label current-action task across 269 held-out rounds.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03731","pdf":"https://arxiv.org/pdf/2608.03731","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03731"},"evidence":{"snippet":"We introduce CARE-Bench, a source-grounded benchmark that evaluates sequential patient-facing triage as a four-label per-turn current-action task.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03731"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CARE-Bench is a source-grounded benchmark for patient-facing triage evaluation with 500 cases and 1,059 patient-disclosure prefixes. It evaluates LLMs on a four-label current-action task across 269 held-out rounds.","whyItMatters":"Provides a standardized evaluation for patient-facing triage, enabling comparison of models on action timing and clarification steps, critical for deployment safety.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"77d397efce78f8690a8abaa4ae9851616af81c5b4ba0f97f3ceb2bcc69c4977f"},"motivation":"Patient-facing medical LLMs and agents increasingly answer symptom questions before clinician contact, where the key safety question is what action the user should take next.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03731","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_carebench_48d01c35","familyId":"bmf_1e30cd79e194","name":"CAREBench","oneLine":"CAREBench evaluates language models on upstream child-safety risks with 500 prompts across twelve categories, assessing recognition, refusal, de-escalation, and redirection.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29685","pdf":"https://arxiv.org/pdf/2606.29685","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.29685"},"evidence":{"snippet":"We introduce CAREBench (Child AI Risk Evaluation), a benchmark to assess such upstream child-safety risks in language models.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29685"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"CAREBench evaluates language models on upstream child-safety risks with 500 prompts across twelve categories, assessing recognition, refusal, de-escalation, and redirection.","whyItMatters":"Child-safety evaluation often focuses on explicit material; CAREBench targets earlier risk scenarios, which could help developers identify policy gaps.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a14405e7b74c365e3e3e5754e9b0c440e62e0b07f49a2b27bf41cea2974b3b9d"},"motivation":"How can we evaluate whether frontier AI systems recognize child-safety risks before they escalate into explicit harm?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.29685","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_67516f1abb5720db","familyId":"catalog_family_67516f1abb5720db","name":"CaseLaw v2","oneLine":"Vals AI private question-answer benchmark over Canadian court cases.","description":"Vals AI private question-answer benchmark over Canadian court cases.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/case_law_v2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_67516f1abb5720db"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valscaselawv2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsCaseLawV2","url":"https://benchlm.ai/benchmarks/valscaselawv2","paperUrl":"https://www.vals.ai/benchmarks/case_law_v2","year":"2026","fullName":"Vals CaseLaw v2","format":"Accuracy score","tasks":"Canadian case-law question answering","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_cast-bench_2c1a3e97","familyId":"bmf_ff0b4c73322f","name":"CaST-Bench","oneLine":"CaST-Bench evaluates causal chain-grounded spatio-temporal reasoning in video question answering. It contains 2,066 questions over 1,015 videos with causal chains annotated as temporal segments and bounding-box tracks. The evaluation suite includes metrics for answer correctness and visual evidence grounding.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2605.23216","pdf":"https://arxiv.org/pdf/2605.23216","project":"https://woven-by-toyota.github.io/CaST-Bench","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.23216"},"evidence":{"snippet":"To address this gap, we introduce CaST-Bench, a benchmark for Causal Chain-Grounded Spatio-Temporal Video Reasoning.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.23216"},"ranking":{},"description":"CaST-Bench evaluates causal chain-grounded spatio-temporal reasoning in video question answering. It contains 2,066 questions over 1,015 videos with causal chains annotated as temporal segments and bounding-box tracks. The evaluation suite includes metrics for answer correctness and visual evidence grounding.","whyItMatters":"Existing video QA benchmarks lack fine-grained causal grounding. CaST-Bench provides a protocol to assess whether VLMs can identify and localize causal evidence chains, supporting progress in transparent and reliable video understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"187f0b5a364b82dcc95b805fe66deb2e12a25edae94d121a6f5d50120adedf25"},"motivation":"Cause-and-effect reasoning in video is a significant challenge for Vision-Language Models (VLMs), as it requires going beyond surface-level perception to a deeper understanding of causal mechanisms.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.23216","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Woven by Toyota","organizationType":"company-research-lab","sourceUrl":"https://woven-by-toyota.github.io/CaST-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_caster-bench_6f29d748","familyId":"bmf_e1ba51a6b988","name":"CASTER-Bench","oneLine":"CASTER-Bench evaluates whether user-generated content achieves positive community resonance based on multimodal attributes, using human annotations across diverse categories.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01897","pdf":"https://arxiv.org/pdf/2606.01897","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01897"},"evidence":{"snippet":"To support this task, we release CASTER-Bench, a comprehensive human-annotated benchmark covering diverse UGC categories.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01897"},"ranking":{},"description":"CASTER-Bench evaluates whether user-generated content achieves positive community resonance based on multimodal attributes, using human annotations across diverse categories.","whyItMatters":"Expands quality assessment from visual fidelity to social engagement, providing a task for modeling community resonance rather than just aesthetic quality.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c503eb4b3e77695cb949d4c68ef593465d667da7daf970fa564ed592c26b493b"},"motivation":"Traditional Video Quality Assessment (VQA) focuses narrowly on aesthetic fidelity, overlooking the complex social dynamics that define quality in User-Generated Content (UGC).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01897","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_catchbench_f09872b4","familyId":"bmf_4e634670acb6","name":"CatchBench","oneLine":"Assesses agent failure auditing across PRE, LIVE, and POST information states with seven task contracts covering evidential and Gold-derived diagnostics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":0.75,"links":{"report":"http://arxiv.org/abs/2608.22808v1","pdf":"https://arxiv.org/pdf/2608.22808v1","project":null,"code":"https://github.com/yzhao062/catchbench","data":null,"hfPaper":null},"evidence":{"snippet":"A benchmark number is therefore not interpretable until the process behind its labels is published and tested for the shortcut it may leave.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22808"},"ranking":{"today":{"score":54,"rank":3,"coverage":0.7,"confidence":"Medium"},"30d":{"score":60,"rank":45,"coverage":0.55,"confidence":"Low"},"90d":{"score":49,"rank":199,"coverage":0.55,"confidence":"Low"}},"description":"Assesses agent failure auditing across PRE, LIVE, and POST information states with seven task contracts covering evidential and Gold-derived diagnostics.","whyItMatters":"Unifies previously separate auditing settings under one interface and exposes label-process shortcuts, helping users judge whether benchmark scores reflect reasoning.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"fdae69c8d88d3ee5e1705f7dff2dbab798cc9177df8a99c01f974b5efe1b3b27"},"motivation":"When can an agent failure be caught?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with released code and data, seven task contracts with defined metrics, and a clear public reuse path via GitHub.","canonicalNameSource":"abstract","canonicalNameEvidence":"CatchBench therefore puts one auditor's question to three information states"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22808v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"Broad agent-auditing scope and a detailed released repository suggest moderate initial visibility, though the niche multi-state framing may limit rapid spread."},"evaluationMode":"public_reusable","publishers":[{"name":"CatchBench maintainers","organizationType":"academic-lab","sourceUrl":"https://github.com/yzhao062/catchbench","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_causal-plan-bench_5fd0e720","familyId":"bmf_d4b6f2500fcb","name":"Causal-Plan-Bench","oneLine":"Causal-Plan-Bench evaluates embodied planning across four causal dimensions (executability, effects, composition, robustness) using 1,200 instances from 12 tasks, with MCQ and rubric-based scoring.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Planning"],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01810","pdf":"https://arxiv.org/pdf/2606.01810","project":null,"code":"https://github.com/THUSI-Lab/Causal-Reasoner","data":null,"hfPaper":"https://huggingface.co/papers/2606.01810"},"evidence":{"snippet":"To this end, we introduce Causal-Plan-Bench, a high-fidelity diagnostic suite curated through multi-stage verification to evaluate embodied planning across four causal dimensions.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01810"},"ranking":{},"description":"Causal-Plan-Bench evaluates embodied planning across four causal dimensions (executability, effects, composition, robustness) using 1,200 instances from 12 tasks, with MCQ and rubric-based scoring.","whyItMatters":"Targets the gap where benchmarks reward linguistic prediction over physical causal reasoning, offering a diagnostic protocol to assess genuine physical agency and support training for grounded planning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b360f841ae56a26306a2e458c474a28dfaf6d4539e212bdce2b7429ced9bd3f6"},"motivation":"Current benchmarks for embodied vision-language planning often favor linguistic next-token prediction over physically grounded next-state reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01810","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"THUSI-Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/THUSI-Lab/Causal-Reasoner","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_causalds_e9c95fdf","familyId":"bmf_c55169b2a8eb","name":"CausalDS","oneLine":"Evaluates causal reasoning in data-science workflows using synthetic scenes with hidden structural causal models, covering Pearl's three rungs and abstention scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.08093","pdf":"https://arxiv.org/pdf/2607.08093","project":null,"code":"https://github.com/andleb/causalds","data":null,"hfPaper":"https://huggingface.co/papers/2607.08093"},"evidence":{"snippet":"We introduce CausalDS, a benchmark for evaluating causal reasoning in agentic data-science workflows.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":"2026-07-10T00:00:00.000Z","githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.08093"},"ranking":{"90d":{"score":28,"rank":253,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates causal reasoning in data-science workflows using synthetic scenes with hidden structural causal models, covering Pearl's three rungs and abstention scoring.","whyItMatters":"Bridges symbolic causal reasoning and realistic data analysis, providing a joint evaluation of reasoning, tool use, and uncertainty quantification in agentic settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"63fa98bc91ff1b80f07a3208683bc0c8a0b01b2787fa430a529bfa0994787fbf"},"motivation":"Large language models (LLMs) increasingly act as integrated data-science agents, combining abstract reasoning with advanced tool use.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.08093","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_causalgame_db1d5cd5","familyId":"bmf_533dd8b8578b","name":"CausalGame","oneLine":"CausalGame is a benchmark for evaluating causal thinking in LLM agents through interactive games. It includes 14 scenarios with selection bias, measurement error, and hidden confounders. Agents design experimental protocols, collect data, and produce solutions with explanations, scored against analytical optima and causal-reasoning rubrics.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.04293","pdf":"https://arxiv.org/pdf/2607.04293","project":"https://causalgame.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.04293"},"evidence":{"snippet":"To this end, we present CausalGame, a benchmark that evaluates the causal thinking capabilities of LLM agents through interactive games.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.04293"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CausalGame is a benchmark for evaluating causal thinking in LLM agents through interactive games. It includes 14 scenarios with selection bias, measurement error, and hidden confounders. Agents design experimental protocols, collect data, and produce solutions with explanations, scored against analytical optima and causal-reasoning rubrics.","whyItMatters":"Existing AI Scientist benchmarks do not isolate causal reasoning under realistic biases. CausalGame provides a structured evaluation for this capability, offering decision value for developers of autonomous research agents seeking to assess robustness against confounders and biases.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9c8fa6ded882bcecb48d786209c9442b6ea865d7ac763f993e6a4de6b33a4acb"},"motivation":"Building AI Scientist agents with Large Language Models (LLMs) has recently attracted growing attention.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"Forty-Third International Conference on Machine Learning (ICML) 2026 as an Oral presentation","evidence":"Zhenhao, Yongqiang, and Chenxi contributed equally to the project. A short version is accepted at the Forty-Third International Conference on Machine Learning (ICML) 2026 as an Oral presentation. Project website https://causalgame.github.io/","evidenceUrl":"https://arxiv.org/abs/2607.04293","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"Forty-Third International Conference on Machine Learning (ICML) 2026 as an Oral presentation","reviewStatus":"accepted","decisionRaw":"Zhenhao, Yongqiang, and Chenxi contributed equally to the project. A short version is accepted at the Forty-Third International Conference on Machine Learning (ICML) 2026 as an Oral presentation. Project website https://causalgame.github.io/","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.04293","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Zhenhao, Yongqiang, and Chenxi contributed equally to the project. A short version is accepted at the Forty-Third International Conference on Machine Learning (ICML) 2026 as an Oral presentation. Project website https://causalgame.github.io/","level":"author-claim"}]}],"publishers":[{"name":"CausalGame Team","organizationType":"academic-lab","sourceUrl":"https://causalgame.github.io/","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_d6fab11018e87c68","familyId":"catalog_family_d6fab11018e87c68","name":"CBNSL","oneLine":"Curriculum Learning of Bayesian Network Structures (CBNSL) benchmark for evaluating algorithms that learn Bayesian network structures from data using curriculum learning techniques. The benchmark uses networks from the bnlearn repository and evaluates structure learning performance using BDeu scoring metrics.","description":"Curriculum Learning of Bayesian Network Structures (CBNSL) benchmark for evaluating algorithms that learn Bayesian network structures from data using curriculum learning techniques. The benchmark uses networks from the bnlearn repository and evaluates structure learning performance using BDeu scoring metrics.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cbnsl","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d6fab11018e87c68"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cbnsl"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cbnsl","url":"https://llm-stats.com/benchmarks/cbnsl","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_cbx-bench_5fdbe064","familyId":"bmf_825405358cac","name":"CBX-Bench","oneLine":"CBX-Bench is a benchmark for quantitatively measuring the quality of Concept Bottleneck Model (CBM) explanations. It uses a council of five open-weight multimodal LLMs to score explanations given an image and class, validated against human preferences. The benchmark maintains a leaderboard for CBM explanation quality.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15404","pdf":"https://arxiv.org/pdf/2608.15404","project":null,"code":"https://github.com/meric-karadag/cbx-bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.15404"},"evidence":{"snippet":"Building on this validated council, we introduce CBX-Bench, a public benchmark and leaderboard: authors of new CBMs can submit their model's explanations, and CBX-Bench scores them with the council and maintains dataset-level rankings of explanation quality.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15404"},"ranking":{"30d":{"score":36,"rank":60,"coverage":0.55,"confidence":"Low"},"90d":{"score":35,"rank":198,"coverage":0.55,"confidence":"Low"}},"description":"CBX-Bench is a benchmark for quantitatively measuring the quality of Concept Bottleneck Model (CBM) explanations. It uses a council of five open-weight multimodal LLMs to score explanations given an image and class, validated against human preferences. The benchmark maintains a leaderboard for CBM explanation quality.","whyItMatters":"CBM interpretability is often evaluated by downstream accuracy, lacking quantitative measures of explanation quality. CBX-Bench offers a human-aligned, scalable evaluation that does not require concept ground truth, enabling comparison of explanation quality across different CBMs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"effde83cc5955e30102b6fbc7361e7b38ab5ce99cfa0bc1f196823f7771a4991"},"motivation":"Concept Bottleneck Models (CBMs) are designed to make visual classification interpretable by expressing predictions through human-understandable concepts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"Explainable Computer Vision (eXCV) Workshop at ECCV 2026","evidence":"Accepted to the Explainable Computer Vision (eXCV) Workshop at ECCV 2026","evidenceUrl":"https://arxiv.org/abs/2608.15404","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"Explainable Computer Vision (eXCV) Workshop at ECCV 2026","reviewStatus":"accepted","decisionRaw":"Accepted to the Explainable Computer Vision (eXCV) Workshop at ECCV 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.15404","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to the Explainable Computer Vision (eXCV) Workshop at ECCV 2026","level":"author-claim"}]}],"publishers":[{"name":"CBX-Bench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/meric-karadag/cbx-bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_4ad13fc4bbc17737","familyId":"catalog_family_4ad13fc4bbc17737","name":"CC-Bench-V2 Backend","oneLine":"CC-Bench-V2 Backend evaluates coding agents on backend development tasks, measuring practical engineering ability to implement server-side logic, APIs, and system components.","description":"CC-Bench-V2 Backend evaluates coding agents on backend development tasks, measuring practical engineering ability to implement server-side logic, APIs, and system components.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cc-bench-v2-backend","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4ad13fc4bbc17737"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cc-bench-v2-backend"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cc-bench-v2-backend","url":"https://llm-stats.com/benchmarks/cc-bench-v2-backend","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_f4ff6092bc298091","familyId":"catalog_family_f4ff6092bc298091","name":"CC-Bench-V2 Frontend","oneLine":"CC-Bench-V2 Frontend evaluates coding agents on frontend development tasks, measuring ability to build UI components, handle styling, and implement client-side logic.","description":"CC-Bench-V2 Frontend evaluates coding agents on frontend development tasks, measuring ability to build UI components, handle styling, and implement client-side logic.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cc-bench-v2-frontend","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f4ff6092bc298091"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cc-bench-v2-frontend"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cc-bench-v2-frontend","url":"https://llm-stats.com/benchmarks/cc-bench-v2-frontend","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_30468f05c76e5998","familyId":"catalog_family_30468f05c76e5998","name":"CC-Bench-V2 Repo Exploration","oneLine":"CC-Bench-V2 Repo Exploration evaluates coding agents on repository-level understanding and navigation, measuring ability to explore, comprehend, and work across entire codebases.","description":"CC-Bench-V2 Repo Exploration evaluates coding agents on repository-level understanding and navigation, measuring ability to explore, comprehend, and work across entire codebases.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cc-bench-v2-repo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_30468f05c76e5998"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cc-bench-v2-repo"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cc-bench-v2-repo","url":"https://llm-stats.com/benchmarks/cc-bench-v2-repo","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_47c588cfffae91bf","familyId":"catalog_family_47c588cfffae91bf","name":"CC-OCR","oneLine":"A comprehensive OCR benchmark for evaluating Large Multimodal Models (LMMs) in literacy. Comprises four OCR-centric tracks: multi-scene text reading, multilingual text reading, document parsing, and key information extraction. Contains 39 subsets with 7,058 fully annotated images, 41% sourced from real applications. Tests capabilities including text grounding, multi-orientation text recognition, and detecting hallucination/repetition across diverse visual challenges.","description":"A comprehensive OCR benchmark for evaluating Large Multimodal Models (LMMs) in literacy. Comprises four OCR-centric tracks: multi-scene text reading, multilingual text reading, document parsing, and key information extraction. Contains 39 subsets with 7,058 fully annotated images, 41% sourced from real applications. Tests capabilities including text grounding, multi-orientation text recognition, and detecting hallucination/repetition across diverse visual challenges.","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Structured Output","Text-To-Image","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_47c588cfffae91bf"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/ccocr"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cc-ocr"}],"catalogSources":[{"catalog":"benchlm","sourceId":"ccOcr","url":"https://benchlm.ai/benchmarks/ccocr","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"CC-OCR","format":"Text extraction from images and documents","tasks":"Optical character recognition","successorKey":null},{"catalog":"llm-stats","sourceId":"cc-ocr","url":"https://llm-stats.com/benchmarks/cc-ocr","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","structured output","text-to-image","vision"],"catalogModelCount":18,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general"},{"id":"bm_cdaf_17e7d375","familyId":"bmf_6378699f7848","name":"CDAF — Cached Descriptive Asset Files","oneLine":"Benchmark compares sidecar-based video understanding against direct analysis on accuracy, prompt token usage, and latency using a reproducible setup with Gemini.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-26","firstSeenAt":"2026-08-27","recognitionConfidence":0.4,"links":{"report":"https://github.com/UditAkhourii/cdaf","pdf":null,"project":"https://zenodo.org/records/22110594","code":"https://github.com/UditAkhourii/cdaf","data":null,"hfPaper":null},"evidence":{"snippet":"Spec, CLI, agent skill, reproducible benchmark.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":101,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:uditakhourii/cdaf"},"ranking":{"30d":{"score":60,"rank":10,"coverage":0.55,"confidence":"Low"},"90d":{"score":57,"rank":29,"coverage":0.55,"confidence":"Low"}},"description":"Benchmark compares sidecar-based video understanding against direct analysis on accuracy, prompt token usage, and latency using a reproducible setup with Gemini.","whyItMatters":"Provides an evaluation contract for AI video workflows to measure efficiency gains from cached sidecar descriptions.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"50a9d69696dabd812750af18a2293656e52b57fb7045ac34d89e5bdb108fd0f0"},"motivation":"cdaf CDAF (Cached Descriptive Asset Files) - open sidecar format for video so AI agents stop re-analyzing the same footage.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:12:36.511123Z","model":"deepseek-v4-pro","decisionReason":"The repository includes a reproducible benchmark with defined metrics and code, and the CDAF format is named in the README.","canonicalNameSource":"official_readme","canonicalNameEvidence":"# CDAF — Cached Descriptive Asset Files"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/UditAkhourii/cdaf","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:16.976372Z"},"attentionForecast":{"score":40,"confidence":"Low","horizon":"7d","reason":"The performance improvement claims are notable but the repository lacks a formal paper release, limiting broader attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_cdr-bench_40384718","familyId":"bmf_11d5499469aa","name":"CDR-Bench","oneLine":"CDR-Bench is a benchmark of 3,462 tasks for evaluating large language models on faithful execution of compositional, order-sensitive data refinement recipes across four domains and 29 operators, with deterministic reference outputs for exact evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.31435","pdf":"https://arxiv.org/pdf/2606.31435","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.31435"},"evidence":{"snippet":"To fill this gap, we introduce CDR-Bench, a comprehensive benchmark featuring 3,462 high-quality tasks spanning four real-world data refinement domains and 29 distinct operators.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31435"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"CDR-Bench is a benchmark of 3,462 tasks for evaluating large language models on faithful execution of compositional, order-sensitive data refinement recipes across four domains and 29 operators, with deterministic reference outputs for exact evaluation.","whyItMatters":"Existing benchmarks leave unclear whether LLMs can directly execute multi-step, order-sensitive data refinement tasks; this benchmark provides a reusable, deterministic evaluation protocol to assess procedural faithfulness in compositional text processing.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b1a757d03ed27beb6ebe96b3552827493948e0a11eda8b9d3c56bfe6770ea86b"},"motivation":"Data refinement involves executing multi-step recipes over evolving text states, where both composition and execution order of processing operators determine the outcome.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31435","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ceo-bench_5cae7f96","familyId":"bmf_a0f9a9f93e4d","name":"CEO-Bench","oneLine":"CEO-Bench evaluates long-horizon agent capabilities by simulating a startup over 500 days. Agents manage pricing, marketing, budgeting, and other business aspects through a programmable Python interface, facing noisy data and changing market conditions. Performance is measured by final company balance against a rule-based baseline.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18543","pdf":"https://arxiv.org/pdf/2606.18543","project":null,"code":"https://github.com/zlab-princeton/ceobench-src","data":null,"hfPaper":"https://huggingface.co/papers/2606.18543"},"evidence":{"snippet":"We introduce CEO-Bench, which evaluates these capabilities together by simulating a representative real-world task: operating a startup for 500 days.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":"2026-06-18T00:00:00.000Z","githubStars":71,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18543"},"ranking":{"90d":{"score":53,"rank":46,"coverage":0.7,"confidence":"Medium"}},"description":"CEO-Bench evaluates long-horizon agent capabilities by simulating a startup over 500 days. Agents manage pricing, marketing, budgeting, and other business aspects through a programmable Python interface, facing noisy data and changing market conditions. Performance is measured by final company balance against a rule-based baseline.","whyItMatters":"CEO-Bench addresses the evaluation gap for agents that must sustain adaptive progress over long horizons, combining uncertainty, information acquisition, and multi-step coordination. It provides a decision-useful benchmark for comparing models on realistic, dynamic business management tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1b6ad59d53ae5b54238321899dfddab40ec687f0a7d5af43c768e1809f4a5b13"},"motivation":"Language model agents are becoming proficient executors at isolated, short-horizon tasks such as software engineering and customer service.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18543","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Princeton University","organizationType":"academic-lab","sourceUrl":"https://github.com/zlab-princeton/ceobench-src","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_ceo-bench_8522cfe6","familyId":"bmf_a0f9a9f93e4d","name":"CEO-Bench","oneLine":"CEO-Bench evaluates LLM agents on strategic resource allocation in multi-round organizational simulations with conflicting advisor inputs.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.17459","pdf":"https://arxiv.org/pdf/2606.17459","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.17459"},"evidence":{"snippet":"We introduce \\textsc{CEO-Bench}, a multi-agent benchmark that evaluates LLMs on CEO-level strategic resource reallocation -- the process of redirecting capital across business units in a multi-round, constraint-rich organizational environment.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.17459"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CEO-Bench evaluates LLM agents on strategic resource allocation in multi-round organizational simulations with conflicting advisor inputs.","whyItMatters":"It probes executive decision-making capabilities beyond isolated cognitive tasks, revealing tradeoffs in agent behavior.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ea7bd26324aaef54d6f5707b8237bae89164ce5c426f4d18250dae4f1ea69759"},"motivation":"Evaluating the decision-making capabilities of large language models (LLMs) is a growing research priority, yet existing benchmarks focus on isolated cognitive tasks such as reasoning, knowledge retrieval, and economic rationality in stylized settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.17459","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_cfagentbench_9ccf4efc","familyId":"bmf_b402658f28ba","name":"CFAgentBench","oneLine":"CFAgentBench is an executable environment for autonomous construction-finance agents, with 1,014 task specifications across 8 domains and 77 families. A subset of 40 tasks (54 with PM extension) has oracle-validated evaluators. Grading uses state diffs, forbidden-side-effect checks, and required-output regexes, with an LLM judge only for reply quality. A public split of 711 tasks is available, and a private split of 303 is reserved for remote scoring.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22000","pdf":"https://arxiv.org/pdf/2606.22000","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.22000"},"evidence":{"snippet":"We introduce CFAgentBench, a reproducible, self-hostable environment and benchmark for autonomous construction-finance agents: a CFO/controller-class agent operating across the real software stack a US construction finance team runs - ERP, project management, email, documents, pay applications, payroll, certified payroll, lien waivers, and bank/treasury portals.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22000"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CFAgentBench is an executable environment for autonomous construction-finance agents, with 1,014 task specifications across 8 domains and 77 families. A subset of 40 tasks (54 with PM extension) has oracle-validated evaluators. Grading uses state diffs, forbidden-side-effect checks, and required-output regexes, with an LLM judge only for reply quality. A public split of 711 tasks is available, and a private split of 303 is reserved for remote scoring.","whyItMatters":"The benchmark addresses the gap in evaluating agents for finance workflows that involve real software stacks and high-stakes transactions. Its focus on functional correctness and a money-movement guard, where correct actions can fail tasks, provides a more realistic measure of deployable competence. The observed performance collapse under repeated attempts highlights the need for reliability assessment beyond single-attempt accuracy.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"714454a11dd9436cd565558f81767e20878f9b940b2058727bb699e86916ff94"},"motivation":"We introduce CFAgentBench, a reproducible, self-hostable environment and benchmark for autonomous construction-finance agents: a CFO/controller-class agent operating across the real software stack a US construction finance team runs - ERP, project management, email, documents, pay applications, payroll, certified payroll, lien waivers, and bank/treasury portals.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22000","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"CFAgentBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.22000","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_27a86d8712b30998","familyId":"catalog_family_27a86d8712b30998","name":"CFEval","oneLine":"CFEval benchmark for evaluating code generation and problem-solving capabilities","description":"CFEval benchmark for evaluating code generation and problem-solving capabilities","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cfeval","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_27a86d8712b30998"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cfeval"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cfeval","url":"https://llm-stats.com/benchmarks/cfeval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_cfm-bench_fc3a1cd0","familyId":"bmf_41e6fe8357e1","name":"CFM-Bench","oneLine":"The evaluation object is unclear from the provided information.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14975","pdf":"https://arxiv.org/pdf/2607.14975","project":"https://www.chaspark.com/\\#/s/CFM-Bench","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.14975"},"evidence":{"snippet":"We release CFM-Bench, a unified multi-domain, multi-task benchmark comprising 157,900 official single-frame examples from six domains spanning 3GPP statistical simulation, two ray-tracing pipelines, terrestrial and aerial measurements, and synchronized vehicular multimodal simulation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14975"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"The evaluation object is unclear from the provided information.","whyItMatters":"The evaluation gap and practical decision value are unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f750cc9655041d308723f72a1ecd0a15296222f7715a5026bf722812bc88441b"},"motivation":"Channel foundation models (CFMs) are commonly evaluated in model-specific pipelines that differ in data, radio configurations, partitions, adaptation procedures, task definitions, and metrics, preventing reproducible comparison across CFMs and against task-specific networks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14975","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_chainswe_b0efa40a","familyId":"bmf_9b5273a86b4c","name":"ChainSWE","oneLine":"ChainSWE evaluates coding agents on sequential, dependent bug fixes within a shared codebase. It includes 304 issues across 54 Python projects, forming chronological chains, and measures performance drop as chain length increases.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.02606","pdf":"https://arxiv.org/pdf/2607.02606","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.02606"},"evidence":{"snippet":"To bridge this gap, we introduce ChainSWE, the first benchmark for evaluating agents on sequential, dependent bug fixes within a shared codebase.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02606"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ChainSWE evaluates coding agents on sequential, dependent bug fixes within a shared codebase. It includes 304 issues across 54 Python projects, forming chronological chains, and measures performance drop as chain length increases.","whyItMatters":"Real-world software maintenance involves streams of related defects, but existing benchmarks evaluate one bug at a time. ChainSWE fills this gap by benchmarking agents on continuous workflows, revealing significant performance degradation on longer chains.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"57a51864fdbc645474213d93abaa4481ad903fabf6bc64fef5231455c5c3f4d1"},"motivation":"Language model (LM) agents are increasingly deployed to maintain codebases over extended periods, fixing streams of related defects while carrying context from one fix to the next.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02606","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_chaosbench-logic_c5cd44be","familyId":"bmf_173654d46d35","name":"ChaosBench-Logic","oneLine":"ChaosBench-Logic v2 evaluates LLM logical reasoning over dynamical systems with 40,886 questions across 165 systems, 27 first-order logic predicates, and 78 axiom edges. The CARE protocol measures calibration and adversarial robustness, reporting metrics such as MCC.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24305","pdf":"https://arxiv.org/pdf/2605.24305","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24305"},"evidence":{"snippet":"We present ChaosBench-Logic v2, a 40,886-question benchmark over 165 dynamical systems with 27 FOL predicates and 78 axiom edges, together with CARE (Calibration- and Adversarial-Robust Evaluation), a protocol that surfaces these pathologies.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24305"},"ranking":{},"description":"ChaosBench-Logic v2 evaluates LLM logical reasoning over dynamical systems with 40,886 questions across 165 systems, 27 first-order logic predicates, and 78 axiom edges. The CARE protocol measures calibration and adversarial robustness, reporting metrics such as MCC.","whyItMatters":"Binary accuracy on reasoning benchmarks hides failures like prior collapse and inconsistency under paraphrase. This benchmark surfaces these pathologies and quantifies reasoning about parameter-dependent dynamics, guiding improvements in LLM reasoning capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9c846a7cbecc884598e92cb84a3b54515adcf6bc3b71d939178af3da7363ec4f"},"motivation":"Standard accuracy on binary reasoning benchmarks hides critical failure modes: prior collapse, inconsistency under paraphrase, and inability to reason about parameter-dependent dynamics.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24305","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_262332403b3cb325","familyId":"catalog_family_262332403b3cb325","name":"CharadesSTA","oneLine":"Charades-STA is a benchmark dataset for temporal activity localization via language queries, extending the Charades dataset with sentence temporal annotations. It contains 12,408 training and 3,720 testing segment-sentence pairs from videos with natural language descriptions and precise temporal boundaries for localizing activities based on language queries.","description":"Charades-STA is a benchmark dataset for temporal activity localization via language queries, extending the Charades dataset with sentence temporal annotations. It contains 12,408 training and 3,720 testing segment-sentence pairs from videos with natural language descriptions and precise temporal boundaries for localizing activities based on language queries.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Multimodal","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/charadessta","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_262332403b3cb325"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/charadessta"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"charadessta","url":"https://llm-stats.com/benchmarks/charadessta","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","multimodal","video","vision"],"catalogModelCount":12,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_chartact_f82f233a","familyId":"bmf_2274e77eb4da","name":"ChartAct","oneLine":"Interactive benchmark for dynamic chart understanding requiring GUI actions to obtain evidence, with 673 charts and 1,440 QA pairs.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26994","pdf":"https://arxiv.org/pdf/2605.26994","project":null,"code":"https://github.com/wulin-wulin/OSWorld_Chart","data":null,"hfPaper":"https://huggingface.co/papers/2605.26994"},"evidence":{"snippet":"To evaluate this ability, we propose ChartAct, an interactive benchmark for dynamic chart understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":10,"hfDailySubmittedAt":null,"githubStars":13,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26994"},"ranking":{},"description":"Interactive benchmark for dynamic chart understanding requiring GUI actions to obtain evidence, with 673 charts and 1,440 QA pairs.","whyItMatters":"Existing chart benchmarks are static; ChartAct evaluates models on real interactive environments, crucial for real-world chart use.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b0f8226e8171cce7d8050140b4184f3150f85bda90610f55179e81ed9e1af0f4"},"motivation":"Charts are widely used to present complex data for analysis and decision making.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26994","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ChartAct Team","organizationType":"academic-lab","sourceUrl":"https://github.com/wulin-wulin/OSWorld_Chart","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_chartanno_e949f852","familyId":"bmf_41b04716eec5","name":"ChartAnno","oneLine":"ChartAnno is a benchmark for evaluating MLLMs on chart annotation generation with 1,200 real-world charts and paired code/instructions. It evaluates annotation quality under different instruction specificity and input settings.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03464","pdf":"https://arxiv.org/pdf/2608.03464","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03464"},"evidence":{"snippet":"To address this gap, we introduce ChartAnno, a benchmark for evaluating MLLMs on chart annotation generation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03464"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ChartAnno is a benchmark for evaluating MLLMs on chart annotation generation with 1,200 real-world charts and paired code/instructions. It evaluates annotation quality under different instruction specificity and input settings.","whyItMatters":"Addresses the underexplored task of chart annotation, providing a testbed for comparing MLLM capabilities in generating communicative chart annotations.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fdc790effa5e4f1c49529a75ccc99adabe8ff5558f229a21857d2d4ae760c553"},"motivation":"Multimodal large language models (MLLMs) have made significant progress in chart understanding, generation, and editing, but their ability to annotate existing charts remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03464","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_chartarena_4eca7a75","familyId":"bmf_9262a71bdf07","name":"ChartArena","oneLine":"ChartArena is a bilingual benchmark for chart parsing covering eight chart families (bar, line, pie, radar, box plot, combination, flowchart, mind map) across three visual scenarios (digital, printed, hand-drawn). It uses a format-agnostic evaluation protocol with structure-aware metrics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01348","pdf":"https://arxiv.org/pdf/2606.01348","project":null,"code":"https://github.com/pspdada/ChartArena","data":null,"hfPaper":"https://huggingface.co/papers/2606.01348"},"evidence":{"snippet":"To address these issues, we introduce ChartArena, a comprehensive bilingual benchmark covering eight chart families spanning both numeric charts and diagrammatic structures, each evaluated across three visual scenarios: digital renderings, printed photos, and hand-drawn photos.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":8,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01348"},"ranking":{},"description":"ChartArena is a bilingual benchmark for chart parsing covering eight chart families (bar, line, pie, radar, box plot, combination, flowchart, mind map) across three visual scenarios (digital, printed, hand-drawn). It uses a format-agnostic evaluation protocol with structure-aware metrics.","whyItMatters":"Existing chart benchmarks cover limited chart types and ignore diagrammatic structures and real-world images. ChartArena provides a unified evaluation across diverse charts and scenarios, enabling assessment of model generalization and identifying capability gaps.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ee977f4e45f8392a61f636322d496194ccad63a5cc654c053fe77f70975d5831"},"motivation":"Charts are a primary medium for conveying quantitative and relational information, yet systematically evaluating chart parsing models remains difficult.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01348","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ChartArena Team","organizationType":"community","sourceUrl":"https://github.com/pspdada/ChartArena","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_85df86ec25b07e1b","familyId":"catalog_family_85df86ec25b07e1b","name":"ChartMuseum","oneLine":"ChartMuseum is a chart question-answering benchmark of 1,162 expert-annotated questions over real-world chart images drawn from 184 sources, including academic figures, infographics, and unconventional chart designs. It specifically targets questions that require visual reasoning, such as comparing unlabeled visual elements, tracking trajectories, and judging spatial relationships.","description":"ChartMuseum is a chart question-answering benchmark of 1,162 expert-annotated questions over real-world chart images drawn from 184 sources, including academic figures, infographics, and unconventional chart designs. It specifically targets questions that require visual reasoning, such as comparing unlabeled visual elements, tracking trajectories, and judging spatial relationships.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/chartmuseum","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_85df86ec25b07e1b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/chartmuseum"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"chartmuseum","url":"https://llm-stats.com/benchmarks/chartmuseum","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_chartography_26c993c6","familyId":"bmf_ac7027e150a2","name":"Chartography","oneLine":"Chartography evaluates vision-language models on 100 chart-understanding tasks sourced from professional practice, with questions authored by domain experts and triple-verified. The benchmark uses pass@1 accuracy as the scoring metric across 30 frontier model configurations.","area":"Vision & 3D","applicationDomains":["Industrial & Engineering","Finance & Economics"],"primaryDomain":"Industrial & Engineering","industrySectors":["Manufacturing","Financial Services"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.10677","pdf":"https://arxiv.org/pdf/2608.10677","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.10677"},"evidence":{"snippet":"We introduce Chartography, a benchmark of 100 tasks that pair charts drawn from professional practice, in domain-specific formats that standard chart benchmarks rarely include, with questions written by professionals who read these charts for a living and independently verified by three additional experts.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10677"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Chartography evaluates vision-language models on 100 chart-understanding tasks sourced from professional practice, with questions authored by domain experts and triple-verified. The benchmark uses pass@1 accuracy as the scoring metric across 30 frontier model configurations.","whyItMatters":"Existing chart benchmarks are saturated and skewed toward simple formats, leaving a gap in measuring performance on realistic, domain-specific charts. Chartography provides a harder, expert-validated test that can differentiate model capabilities in high-stakes professional contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"596c35b8826016fed1a9d257e3e85883bbd76f1636a65e7969490e6c422c1f50"},"motivation":"Professionals across medicine, engineering, finance, manufacturing, and the sciences often make consequential decisions from charts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"2nd Workshop on Benchmarking Evidence-Aligned Multimodal Reasoning (BEAM 2), ECCV 2026","evidence":"16 pages, 5 figures, 5 tables. Accepted at the 2nd Workshop on Benchmarking Evidence-Aligned Multimodal Reasoning (BEAM 2), ECCV 2026","evidenceUrl":"https://arxiv.org/abs/2608.10677","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"2nd Workshop on Benchmarking Evidence-Aligned Multimodal Reasoning (BEAM 2), ECCV 2026","reviewStatus":"accepted","decisionRaw":"16 pages, 5 figures, 5 tables. Accepted at the 2nd Workshop on Benchmarking Evidence-Aligned Multimodal Reasoning (BEAM 2), ECCV 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.10677","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"16 pages, 5 figures, 5 tables. Accepted at the 2nd Workshop on Benchmarking Evidence-Aligned Multimodal Reasoning (BEAM 2), ECCV 2026","level":"author-claim"}]}],"publishers":[{"name":"Chartography Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.10677","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"cross-domain","catalogSources":[{"catalog":"llm-stats","sourceId":"chartography","url":"https://llm-stats.com/benchmarks/chartography","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0},{"id":"catalog_041bdfe4c841373e","familyId":"catalog_family_041bdfe4c841373e","name":"Chartography (no tools)","oneLine":"Professional chart understanding across 100 specialized chart types with expert-set answer tolerances.","description":"Professional chart understanding across 100 specialized chart types with expert-set answer tolerances.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://surgehq.ai/blog/chartography","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_041bdfe4c841373e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/chartography"}],"catalogSources":[{"catalog":"benchlm","sourceId":"chartography","url":"https://benchlm.ai/benchmarks/chartography","paperUrl":"https://surgehq.ai/blog/chartography","year":"2026","fullName":"Chartography without tools","format":"Accuracy without tools","tasks":"100 specialized chart types","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_56d50855b2e0768e","familyId":"catalog_family_56d50855b2e0768e","name":"Chartography (tools)","oneLine":"Professional chart understanding with a container, standard libraries, and image cropping.","description":"Professional chart understanding with a container, standard libraries, and image cropping.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://surgehq.ai/blog/chartography","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_56d50855b2e0768e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/chartographywithtools"}],"catalogSources":[{"catalog":"benchlm","sourceId":"chartographyWithTools","url":"https://benchlm.ai/benchmarks/chartographywithtools","paperUrl":"https://surgehq.ai/blog/chartography","year":"2026","fullName":"Chartography with image and code tools","format":"Accuracy with tools","tasks":"100 specialized chart types","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0a5d08ac3f6ebcb5","familyId":"catalog_family_0a5d08ac3f6ebcb5","name":"ChartQA","oneLine":"ChartQA is a large-scale benchmark comprising 9.6K human-written questions and 23.1K questions generated from human-written chart summaries, designed to evaluate models' abilities in visual and logical reasoning over charts.","description":"ChartQA is a large-scale benchmark comprising 9.6K human-written questions and 23.1K questions generated from human-written chart summaries, designed to evaluate models' abilities in visual and logical reasoning over charts.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/chartqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0a5d08ac3f6ebcb5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/chartqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"chartqa","url":"https://llm-stats.com/benchmarks/chartqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":26,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_76c1795b33807a83","familyId":"catalog_family_76c1795b33807a83","name":"ChartQAPro","oneLine":"ChartQAPro is a challenging benchmark for question answering over diverse, real-world charts and infographics.","description":"ChartQAPro is a challenging benchmark for question answering over diverse, real-world charts and infographics.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/chartqapro","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_76c1795b33807a83"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/chartqapro"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"chartqapro","url":"https://llm-stats.com/benchmarks/chartqapro","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_chartsync_6261046a","familyId":"bmf_700e28d4e4ca","name":"ChartSync","oneLine":"ChartSync evaluates image editing models on chart editing tasks, including text-only edits and visuo-logical cascading edits (VLCE) requiring synchronized text and geometry changes. The benchmark contains 870 expert-validated triplets across nine chart families, with objective visual metrics and a vision-language model judge for evaluation.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.10301","pdf":"https://arxiv.org/pdf/2607.10301","project":null,"code":"https://github.com/kaka-yjk/ChartSyncCodebase","data":null,"hfPaper":"https://huggingface.co/papers/2607.10301"},"evidence":{"snippet":"To systematically evaluate this capability, we introduce ChartSync, an expert-validated benchmark constructed via a programmatic rendering pipeline that guarantees deterministic visuo-logical coupling for the ground truth.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.10301"},"ranking":{"90d":{"score":30,"rank":245,"coverage":0.7,"confidence":"Medium"}},"description":"ChartSync evaluates image editing models on chart editing tasks, including text-only edits and visuo-logical cascading edits (VLCE) requiring synchronized text and geometry changes. The benchmark contains 870 expert-validated triplets across nine chart families, with objective visual metrics and a vision-language model judge for evaluation.","whyItMatters":"Existing image editing benchmarks often overlook structured data charts where data modifications require geometric synchronization. ChartSync provides a reproducible evaluation to assess models' ability to handle dependency-aware cascading updates, revealing capability gaps in open-source versus proprietary models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"353834346449c1c354ef5316925b669a576cb6aeb0cd776ee64d32ca80034f5b"},"motivation":"Generative image editing models struggle with structured statistical charts when data modifications require geometric synchronization.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.10301","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ChartSync Team","organizationType":"academic-lab","sourceUrl":"https://github.com/kaka-yjk/ChartSyncCodebase","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"lib_charxiv","familyId":"family_charxiv","name":"CharXiv","oneLine":"Established benchmark family · Chart Understanding.","area":"Chart Understanding","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Chart Understanding"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2406.18521","pdf":null,"project":"https://charxiv.github.io/","code":"https://github.com/princeton-nlp/CharXiv","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_charxiv"},"ranking":{},"recordType":"family","aliases":["Charting Research on Chart Understanding"],"sourceAttribution":[{"role":"official-project","url":"https://charxiv.github.io/"}],"adoptionRefs":["google-gemini25"],"modelReportReferences":[{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"charxiv","url":"https://benchlm.ai/benchmarks/charxiv","paperUrl":"https://charxiv.github.io/","year":"2024","fullName":"CharXiv Reasoning","format":"Chart understanding and reasoning","tasks":"Scientific chart reasoning","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0},{"id":"catalog_3beb0a3e416e63b1","familyId":"catalog_family_3beb0a3e416e63b1","name":"CharXiv w/o tools","oneLine":"Tool-free variant of CharXiv that isolates raw visual reasoning ability without code execution or tool augmentation.","description":"Tool-free variant of CharXiv that isolates raw visual reasoning ability without code execution or tool augmentation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://charxiv.github.io/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3beb0a3e416e63b1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/charxivnotools"}],"catalogSources":[{"catalog":"benchlm","sourceId":"charxivNoTools","url":"https://benchlm.ai/benchmarks/charxivnotools","paperUrl":"https://charxiv.github.io/","year":"2024","fullName":"CharXiv Reasoning without tools","format":"Chart understanding without tools","tasks":"Scientific chart reasoning (tool-free)","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_f33650566edd9798","familyId":"catalog_family_f33650566edd9798","name":"CharXiv-D","oneLine":"CharXiv-D is the descriptive questions subset of the CharXiv benchmark, designed to assess multimodal large language models' ability to extract basic information from scientific charts. It contains descriptive questions covering information extraction, enumeration, pattern recognition, and counting across 2,323 diverse charts from arXiv papers, all curated and verified by human experts.","description":"CharXiv-D is the descriptive questions subset of the CharXiv benchmark, designed to assess multimodal large language models' ability to extract basic information from scientific charts. It contains descriptive questions covering information extraction, enumeration, pattern recognition, and counting across 2,323 diverse charts from arXiv papers, all curated and verified by human experts.","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Structured Output","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/charxiv-d","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f33650566edd9798"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/charxiv-d"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"charxiv-d","url":"https://llm-stats.com/benchmarks/charxiv-d","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","structured output","vision"],"catalogModelCount":17,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general"},{"id":"catalog_65ac5d8988d5755d","familyId":"catalog_family_65ac5d8988d5755d","name":"CharXiv-R","oneLine":"CharXiv-R is the reasoning component of the CharXiv benchmark, focusing on complex reasoning questions that require synthesizing information across visual chart elements. It evaluates multimodal large language models on their ability to understand and reason about scientific charts from arXiv papers through various reasoning tasks.","description":"CharXiv-R is the reasoning component of the CharXiv benchmark, focusing on complex reasoning questions that require synthesizing information across visual chart elements. It evaluates multimodal large language models on their ability to understand and reason about scientific charts from arXiv papers through various reasoning tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/charxiv-r","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_65ac5d8988d5755d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/charxiv-r"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"charxiv-r","url":"https://llm-stats.com/benchmarks/charxiv-r","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":53,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_8e2830bbb170cb65","familyId":"catalog_family_8e2830bbb170cb65","name":"ChatCVQA","oneLine":"A conversational visual QA benchmark that tests multi-turn grounded answering over images and documents.","description":"A conversational visual QA benchmark that tests multi-turn grounded answering over images and documents.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8e2830bbb170cb65"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/chatcvqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"chatCvqa","url":"https://benchlm.ai/benchmarks/chatcvqa","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"ChatCVQA","format":"Multi-turn image-grounded QA","tasks":"Conversational visual QA","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_chehre_a4cc7955","familyId":"bmf_2ff6b596d710","name":"Chehre","oneLine":"Chehre is a video dataset of 2,111 facial expressions prompted by 40 emojis, with annotations transferred to synthetic faces. It defines two tasks: dominant expression recognition and distributional expression recognition, evaluating models' ability to predict human-rated labels and capture response diversity.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-19","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.21657","pdf":"https://arxiv.org/pdf/2606.21657","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.21657"},"evidence":{"snippet":"We define two benchmark tasks: dominant expression recognition, which tests whether models recover the top human-rated labels, and distributional expression recognition, which tests whether models capture the diversity of human responses.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.21657"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Chehre is a video dataset of 2,111 facial expressions prompted by 40 emojis, with annotations transferred to synthetic faces. It defines two tasks: dominant expression recognition and distributional expression recognition, evaluating models' ability to predict human-rated labels and capture response diversity.","whyItMatters":"Existing facial expression benchmarks rely on static images and basic categories, limiting evaluation of dynamic, diverse expressions. Chehre provides a controlled resource for measuring model performance on varied, distributional perception, with tasks that reveal gaps in current vision-language models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"89e7ffee0b8cf50b07733c26b83e64d5c4a00663fa5fe78fec03596c7235604c"},"motivation":"Facial expressions are nonverbal social signals used in human interaction, but facial expression recognition datasets often focus on static images, basic emotion categories, or single deterministic annotations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.21657","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Chehre Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.21657","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_chem-world_1eaa9499","familyId":"bmf_e29ccb36b6a1","name":"Chem World","oneLine":"Chem World integrates 17 chemical datasets with 800,000+ molecules for property prediction, offering a unified evaluation platform.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.28079","pdf":"https://arxiv.org/pdf/2607.28079","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.28079"},"evidence":{"snippet":"In this work, we introduce Chem World, a comprehensive benchmark for chemical property prediction that integrates 17 diverse chemical datasets with over 800,000 molecular samples, covering various properties including density, electrical conductivity, solubility, and other molecular characteristics.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.28079"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Chem World integrates 17 chemical datasets with 800,000+ molecules for property prediction, offering a unified evaluation platform.","whyItMatters":"It aims to standardize chemical property prediction evaluation, improving reliability and comparison across models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8aef31d0c752c70a4321f2fc42f7baf570478940e7f49798346670f57bfd3bae"},"motivation":"Chemical property prediction plays a critical role in accelerating scientific discovery in chemistry, materials science, and drug development.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.28079","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_chem2gen-bench_dc691880","familyId":"bmf_2b1bf773f6b9","name":"Chem2Gen-Bench","oneLine":"Chem2Gen-Bench evaluates chemical-to-genetic translation using 260,084 chemical and 1,099,045 genetic perturbation profiles in cell-target contexts. It measures pairwise alignment, retrieval success, and representation quality across matched perturbation settings.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Information retrieval"],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-19","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.21109","pdf":"https://arxiv.org/pdf/2606.21109","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.21109"},"evidence":{"snippet":"We introduce Chem2Gen-Bench, a benchmark comprising 260,084 chemical and 1,099,045 genetic perturbation profiles organized into cell-target contexts, and evaluate pairwise alignment, retrieval, protocol covariate associations, feature spaces, and foundation-model embeddings.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.21109"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Chem2Gen-Bench evaluates chemical-to-genetic translation using 260,084 chemical and 1,099,045 genetic perturbation profiles in cell-target contexts. It measures pairwise alignment, retrieval success, and representation quality across matched perturbation settings.","whyItMatters":"Chemical and genetic perturbations are often studied separately, leaving translation between them under-tested. This benchmark provides a comparative evaluation of retrieval and representation methods, offering decision value for selecting models that align perturbations around shared targets.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"da1053bd97aa295d773b634e1439ed27fc1b50f755fd7ef71735a0fb42842f94"},"motivation":"Virtual-cell and perturbation models are increasingly used to predict cellular responses for biomedical discovery, but chemical and genetic perturbations are not automatically interchangeable.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.21109","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"specific"},{"id":"bm_chemcotbench-v2_5360db80","familyId":"bmf_43f17e5151db","name":"ChemCoTBench-V2","oneLine":"ChemCoTBench-V2 evaluates chemical reasoning at the process level, with 5,620 samples across molecular understanding, editing, optimization, and reaction prediction. It checks intermediate steps using deterministic chemistry rules and reference traces.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03660","pdf":"https://arxiv.org/pdf/2606.03660","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03660"},"evidence":{"snippet":"We introduce ChemCoTBench-V2, a rule-verifiable diagnostic benchmark for low-cost, auditable evaluation of structured, verifier-addressable chemical reasoning traces.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03660"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"ChemCoTBench-V2 evaluates chemical reasoning at the process level, with 5,620 samples across molecular understanding, editing, optimization, and reaction prediction. It checks intermediate steps using deterministic chemistry rules and reference traces.","whyItMatters":"Existing chemistry benchmarks only score final answers, missing violations in reasoning. ChemCoTBench-V2 provides auditable, verifiable process-level evaluation with three separate signals, enabling fine-grained model comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c3e8746331d8b2fba17760cb5543ca5052e315212ec565b0c995d37dce291805"},"motivation":"Large language models are increasingly used as chemistry assistants, yet most chemistry benchmarks still score only final answers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03660","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_chess-world-model_85ede047","familyId":"bmf_3f59e56e9de7","name":"Chess-World-Model","oneLine":"Chess-World-Model evaluates state tracking by predicting the exact board state after sequences of legal moves from 10 million chess games, with accuracy as the metric.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30100","pdf":"https://arxiv.org/pdf/2605.30100","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30100"},"evidence":{"snippet":"We introduce Chess-World-Model, a large-scale state-tracking benchmark built from 10 million real chess games, where models predict the exact board state reached after a sequence of legal moves.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30100"},"ranking":{},"description":"Chess-World-Model evaluates state tracking by predicting the exact board state after sequences of legal moves from 10 million chess games, with accuracy as the metric.","whyItMatters":"State tracking is under-tested in realistic domains. This benchmark exposes limitations in Transformer and RNN models that scale alone may hide, offering a practical testbed for world models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e363a61d728c55038df32e934d77df97017119f7a80eb09729a27225c79abe70"},"motivation":"World models require state tracking, which is the ability to maintain a correct latent state across action sequences.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30100","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_db08c7c91811297e","familyId":"catalog_family_db08c7c91811297e","name":"CheXpert CXR","oneLine":"CheXpert is a large dataset of 224,316 chest radiographs from 65,240 patients for automated chest X-ray interpretation. The dataset includes uncertainty labels for 14 medical observations extracted from radiology reports. It serves as a benchmark for developing and evaluating automated chest radiograph interpretation models.","description":"CheXpert is a large dataset of 224,316 chest radiographs from 65,240 patients for automated chest X-ray interpretation. The dataset includes uncertainty labels for 14 medical observations extracted from radiology reports. It serves as a benchmark for developing and evaluating automated chest radiograph interpretation models.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Healthcare","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/chexpert-cxr","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_db08c7c91811297e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/chexpert-cxr"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"chexpert-cxr","url":"https://llm-stats.com/benchmarks/chexpert-cxr","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["healthcare","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_childeval_9775399e","familyId":"bmf_61cce4f98ffa","name":"ChildEval","oneLine":"ChildEval is a benchmark for evaluating LLMs' ability to infer and follow child-centered preferences in long-context conversations. It contains 29K synthesized persona profiles of children aged 3-6, with explicit and implicit preference expressions across five top-level and fourteen sub-level categories.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27805","pdf":"https://arxiv.org/pdf/2605.27805","project":null,"code":"https://github.com/ziyanluo/ChildEval","data":null,"hfPaper":"https://huggingface.co/papers/2605.27805"},"evidence":{"snippet":"To address this gap, we introduce ChildEval, a benchmark for evaluating LLMs' ability to infer and follow child-centered preferences in long-context conversations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27805"},"ranking":{},"description":"ChildEval is a benchmark for evaluating LLMs' ability to infer and follow child-centered preferences in long-context conversations. It contains 29K synthesized persona profiles of children aged 3-6, with explicit and implicit preference expressions across five top-level and fourteen sub-level categories.","whyItMatters":"Personalization for children is under-explored relative to adults. ChildEval provides a protocol to test whether LLMs can infer and follow child-specific preferences, addressing a gap in personalized conversational AI evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"17dc3cabe5cc52157285d91e37e47bd5f0faaa2b978a936491bf9f3f1c02adc8"},"motivation":"While LLMs enable personalized chatbots, their effectiveness in child-centered personalization remains unclear, as systematic evaluation of child-specific preferences is still lacking.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27805","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ChildEval Team","organizationType":"academic-lab","sourceUrl":"https://github.com/ziyanluo/ChildEval","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning","Long Context & Memory"],"domainScope":"general"},{"id":"catalog_2dd1a5fa5242cca8","familyId":"catalog_family_2dd1a5fa5242cca8","name":"Chinese-SimpleQA","oneLine":"A Chinese short-form factuality benchmark reported by DeepSeek for V4 model evaluations.","description":"A Chinese short-form factuality benchmark reported by DeepSeek for V4 model evaluations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2dd1a5fa5242cca8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/chinesesimpleqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"chineseSimpleQa","url":"https://benchlm.ai/benchmarks/chinesesimpleqa","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"Chinese-SimpleQA","format":"Short-form factual QA","tasks":"Chinese factual questions","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_chlogic_3b9dd8db","familyId":"bmf_760dd213a7b3","name":"ChLogic","oneLine":"ChLogic is an English–Chinese aligned benchmark for evaluating logical reasoning robustness across surface realizations. It includes 3,000 general, 2,000 difficult, and 1,500 Chinese-only items derived from formal logical templates, with each aligned item pairing one English reference with five Chinese variants.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Robustness"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.17905","pdf":"https://arxiv.org/pdf/2606.17905","project":null,"code":"https://github.com/0328zpx/ChLogic","data":null,"hfPaper":"https://huggingface.co/papers/2606.17905"},"evidence":{"snippet":"We introduce ChLogic, an English--Chinese aligned benchmark that tests whether models preserve logical reasoning performance when the same latent logical structure is expressed in English and diverse Chinese surface realizations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":"2026-06-17T00:00:00.000Z","githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.17905"},"ranking":{"90d":{"score":26,"rank":293,"coverage":0.7,"confidence":"Medium"}},"description":"ChLogic is an English–Chinese aligned benchmark for evaluating logical reasoning robustness across surface realizations. It includes 3,000 general, 2,000 difficult, and 1,500 Chinese-only items derived from formal logical templates, with each aligned item pairing one English reference with five Chinese variants.","whyItMatters":"Existing logical reasoning benchmarks focus on English, so it is unclear whether models retain performance when the same logical structure is expressed in Chinese. ChLogic provides a cross-lingual stress test, helping identify language-specific gaps and translation artifacts that affect multilingual reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d65f67ca676a6bafd2cc6a30259edab1bf36a755ab003b5a50f60c1c4e3b6182"},"motivation":"Large language models perform increasingly well on standardized logical reasoning benchmarks, but whether this ability remains robust beyond English is unclear.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.17905","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ChLogic team","organizationType":"academic-lab","sourceUrl":"https://github.com/0328zpx/ChLogic","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_choroplethmap-bench_3460d576","familyId":"bmf_92308d9f928f","name":"ChoroplethMap-Bench","oneLine":"ChoroplethMap-Bench evaluates spatial understanding of foundation models with 2,400 synthetic choropleth maps, corresponding GeoJSON data, and 12,000 questions across five cognitive dimensions (Identify, Spatial Recognition, Compare, Rank, Delineate). Models are assessed under Data Only, Map Only, and Data + Map conditions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.17999","pdf":"https://arxiv.org/pdf/2607.17999","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.17999"},"evidence":{"snippet":"We introduce ChoroplethMap-Bench, a controlled benchmark containing 2,400 synthetic choropleth maps, corresponding GeoJSON data, and 12,000 questions across five cognitive dimensions: Identify, Spatial Recognition, Compare, Rank, and Delineate.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.17999"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ChoroplethMap-Bench evaluates spatial understanding of foundation models with 2,400 synthetic choropleth maps, corresponding GeoJSON data, and 12,000 questions across five cognitive dimensions (Identify, Spatial Recognition, Compare, Rank, Delineate). Models are assessed under Data Only, Map Only, and Data + Map conditions.","whyItMatters":"The benchmark assesses whether cartographic representations add value over structured geodata for machine spatial reasoning, addressing a gap in evaluating map-based inputs. It supports decisions on when to incorporate visual map data in geospatial AI systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bcc47b172b981fc992a5061e574108d8fdd67c9465d2fc1200e6ecdf7d15c704"},"motivation":"Spatial understanding is crucial for foundation models (FMs), and maps have long helped humans organize and reason about geographic information.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.17999","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"The authors","organizationType":"community","sourceUrl":"https://arxiv.org/abs/2607.17999","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_chronobench_44b7b76d","familyId":"bmf_384c2b00f08c","name":"ChronoBench","oneLine":"ChronoBench evaluates long-term temporal understanding in remote sensing across four progressive cognitive levels: land cover perception, temporal recognition, long-term memory, and spatio-temporal reasoning. It comprises 12 sub-tasks and 17,689 QA pairs over 3,469 images spanning 500 regions across 39 U.S. cities.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.15768","pdf":"https://arxiv.org/pdf/2607.15768","project":null,"code":"https://github.com/IntelliSensing/GeoChrono","data":null,"hfPaper":"https://huggingface.co/papers/2607.15768"},"evidence":{"snippet":"To fill this gap, we introduce ChronoBench, a multidimensional benchmark that decomposes this task into four progressive cognitive levels (i.e., Land Cover Perception, Temporal Recognition, Long-Term Memory, and Spatio-Temporal Reasoning).","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":9,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.15768"},"ranking":{"90d":{"score":34,"rank":204,"coverage":0.7,"confidence":"Medium"}},"description":"ChronoBench evaluates long-term temporal understanding in remote sensing across four progressive cognitive levels: land cover perception, temporal recognition, long-term memory, and spatio-temporal reasoning. It comprises 12 sub-tasks and 17,689 QA pairs over 3,469 images spanning 500 regions across 39 U.S. cities.","whyItMatters":"Existing remote sensing benchmarks typically focus on static or bi-temporal analysis, lacking a systematic dissection of long-term temporal competencies. ChronoBench provides a multidimensional evaluation that isolates specific cognitive bottlenecks, enabling targeted model improvement. Its integration with lmms-eval supports standardized comparison of multimodal LLMs for satellite image time series.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4209e15b06135587f7b1fe72cd1383fc1680fe72216c0a023a38d9dcacd2c4db"},"motivation":"Remote sensing offers an unparalleled vantage point for observing the Earth's long-term surface evolution, yet it demands that a model not only perceive land cover at isolated moments, but also track changes, memorize evolution histories, and reason across time and space.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ACM MM 2026","evidence":"Accepted to ACM MM 2026","evidenceUrl":"https://arxiv.org/abs/2607.15768","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ACM MM 2026","reviewStatus":"accepted","decisionRaw":"Accepted to ACM MM 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.15768","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to ACM MM 2026","level":"author-claim"}]}],"publishers":[{"name":"IntelliSensing Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/IntelliSensing/GeoChrono","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_chronophybench_c3f107b4","familyId":"bmf_b04565b83dc6","name":"ChronoPhyBench","oneLine":"ChronoPhyBench evaluates multimodal LLMs on chronological physical dynamics reasoning via next-state prediction and VQA, using video frames and captions for single-image selection and multi-frame sorting.","area":"Safety & Trustworthiness","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07962","pdf":"https://arxiv.org/pdf/2606.07962","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07962"},"evidence":{"snippet":"Motivated by this, and to rigorously mitigate language modality bias and shortcuts, we propose a novel multimodal Chrono}logical Physical Dynamics Reasoning Benchmark ChronoPhyBench, which unifies next state prediction with Visual Question Answering (VQA) paradigms by conditioning on historical video context and textual captions to enforce models to deduce subsequent physical states through both single image selection and the inherently more complex task of multiple frame chronological sorting.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07962"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ChronoPhyBench evaluates multimodal LLMs on chronological physical dynamics reasoning via next-state prediction and VQA, using video frames and captions for single-image selection and multi-frame sorting.","whyItMatters":"Addresses the gap of distinguishing true multimodal reasoning from language-prior exploitation, providing a robust framework to measure physical reasoning and hallucination rates in models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5bd1375b6ae9121f7d0eef7ce0cf80d9d83dde07d58b319f355faf630e17b75a"},"motivation":"Recent advancements in Multimodal Large Language Models (MLLMs) have demonstrated remarkable proficiency in open-world reasoning and understanding.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07962","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ChronoPhyBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.07962","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_chronoqg_6a4613e9","familyId":"bmf_ac031db3c5df","name":"ChronoQG","oneLine":"The evaluation object is unclear from the provided information.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14770","pdf":"https://arxiv.org/pdf/2607.14770","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.14770"},"evidence":{"snippet":"We propose ChronoQG, the first temporally expressive and hop-bounded benchmark construction framework for TKGQG.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14770"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"The evaluation object is unclear from the provided information.","whyItMatters":"The evaluation gap and practical decision value are unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"74860b5074cf170a6ce0bc75155bf3358e70cf387aae5becf8c8ce75ce799307"},"motivation":"Knowledge graph question generation (KGQG) aims to generate natural-language questions from structured graph evidence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14770","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_chronostate_2f5e6d2f","familyId":"bmf_ba0e55d44cb0","name":"ChronoState","oneLine":"ChronoState evaluates whether a frozen language model can compose hidden elapsed-time scalars with symbolic task state to select temporal actions, using forced-choice accuracy under direct supervision.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09124","pdf":"https://arxiv.org/pdf/2608.09124","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09124"},"evidence":{"snippet":"We introduce ChronoState, a compositional temporal-state benchmark in which symbolic state appears in the prompt, elapsed seconds tau are supplied through a hidden chronometric-injection channel, and the model selects a forced-choice temporal action.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09124"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ChronoState evaluates whether a frozen language model can compose hidden elapsed-time scalars with symbolic task state to select temporal actions, using forced-choice accuracy under direct supervision.","whyItMatters":"The benchmark investigates a narrow mechanism for injecting time information into LLMs, but the results do not generalize broadly and the setup is primarily for probing one architectural variant.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fe4b7db4c7f15f1f812314917070aee3df98627475821ab3584cb5d3ac38ca89"},"motivation":"Temporal decisions in language-model systems often depend on both symbolic task state and elapsed wall-clock time, such as cache expiration, job completion, quota resets, deadlines, or stale sessions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09124","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_c80d5333e2d792e7","familyId":"catalog_family_c80d5333e2d792e7","name":"CI Memories Coverage","oneLine":"CI Memories measures privacy behavior in memory-enabled agents using contextual-integrity scenarios. This metric reports evaluation coverage.","description":"CI Memories measures privacy behavior in memory-enabled agents using contextual-integrity scenarios. This metric reports evaluation coverage.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Memory","Privacy","Safety","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ci-memories-coverage","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c80d5333e2d792e7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ci-memories-coverage"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ci-memories-coverage","url":"https://llm-stats.com/benchmarks/ci-memories-coverage","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["memory","privacy","safety","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_851343005b4a282e","familyId":"catalog_family_851343005b4a282e","name":"CI Memories Violation Rate","oneLine":"CI Memories measures privacy failures in memory-enabled agents using contextual-integrity scenarios. This metric is the violation rate; lower is better.","description":"CI Memories measures privacy failures in memory-enabled agents using contextual-integrity scenarios. This metric is the violation rate; lower is better.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Memory","Privacy","Safety","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ci-memories-violation","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_851343005b4a282e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ci-memories-violation"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ci-memories-violation","url":"https://llm-stats.com/benchmarks/ci-memories-violation","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["memory","privacy","safety","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_ciaware-bench_e29a610f","familyId":"bmf_78a7e9aa9db3","name":"CIAware-Bench","oneLine":"CIAware-Bench measures control intervention awareness in language models through four task domains: essay writing, BigCodeBench, Bash Arena, and SHADE-Arena. Models are tested on distinguishing their own trajectories from those modified by a control protocol, with variations in watermarking, side-task presence, and protocol.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11063","pdf":"https://arxiv.org/pdf/2606.11063","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.11063"},"evidence":{"snippet":"We introduce \\textbf{CIAware-Bench}, a benchmark for measuring \\textbf{c}ontrol \\textbf{i}ntervention (CI) awareness across frontier models.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.11063"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CIAware-Bench measures control intervention awareness in language models through four task domains: essay writing, BigCodeBench, Bash Arena, and SHADE-Arena. Models are tested on distinguishing their own trajectories from those modified by a control protocol, with variations in watermarking, side-task presence, and protocol.","whyItMatters":"This benchmark addresses the evaluation gap of determining whether models can detect modifications to their trajectories, which is critical for AI control protocols. It provides practical value by informing the design of interventions that are harder for models to detect, thereby improving the robustness of AI oversight systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"69e83acb4b5b3c483e323308f788839786750feff0a7b9062cfa8af73f86a2db"},"motivation":"AI control protocols oversee untrusted models by monitoring their actions and modifying potentially unsafe steps, often using a trusted model.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11063","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_citbench_cd2dd37c","familyId":"bmf_dc5c06b0c4e1","name":"CITBench","oneLine":"CITBench evaluates LLMs on interactive tabular data processing, covering table matching, cleaning, augmentation, and transformation across 18 task types and 1,296 instances.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.DB"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00018","pdf":"https://arxiv.org/pdf/2608.00018","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00018"},"evidence":{"snippet":"To bridge this gap, we introduce CITBench, a comprehensive benchmark for evaluating LLMs on interactive tabular data processing.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00018"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CITBench evaluates LLMs on interactive tabular data processing, covering table matching, cleaning, augmentation, and transformation across 18 task types and 1,296 instances.","whyItMatters":"Tabular data processing benchmarks often focus on single-turn reasoning; interactive multi-turn settings remain underevaluated.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"855d9a5c1cf28d5cb968b2fc8ab2cfffae0647e1632ccd92838e5cc0fcde9a79"},"motivation":"Tabular data processing is central to data work, and LLM-based assistants have recently shown promising capabilities in supporting such tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00018","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_cityrep_fb8ebfd4","familyId":"bmf_443c2445803a","name":"CITYREP","oneLine":"CityRep is a benchmark for urban representation learning across 8 cities and 8 tasks with spatially structured splits to mitigate leakage.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26036","pdf":"https://arxiv.org/pdf/2605.26036","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.26036"},"evidence":{"snippet":"To address this, we propose CityRep, a unified benchmark that evaluates urban representations across data modalities, cities, and tasks using spatially structured splits.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26036"},"ranking":{},"description":"CityRep is a benchmark for urban representation learning across 8 cities and 8 tasks with spatially structured splits to mitigate leakage.","whyItMatters":"It addresses inconsistencies in urban representation evaluation and supports generalization-aware model selection.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"180885379bfe25d2f368eaa9717ca09a77ea342dd56f3ef7affa362a9fa8110a"},"motivation":"Urban representation learning encodes complex urban environments into general-purpose embeddings for diverse downstream tasks and emerging urban foundation models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26036","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_citytrajbench_4320b29c","familyId":"bmf_b181f74b4134","name":"CityTrajBench","oneLine":"CityTrajBench evaluates city-scale vehicle trajectory generation methods with standardized preprocessing, model adaptation, and multi-level evaluation across statistical, VAE, GAN, diffusion, and flow-matching models.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02287","pdf":"https://arxiv.org/pdf/2606.02287","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02287"},"evidence":{"snippet":"To address this issue, we present CityTrajBench, a unified benchmark framework and protocol for city-scale vehicle trajectory generation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02287"},"ranking":{},"description":"CityTrajBench evaluates city-scale vehicle trajectory generation methods with standardized preprocessing, model adaptation, and multi-level evaluation across statistical, VAE, GAN, diffusion, and flow-matching models.","whyItMatters":"Provides a unified protocol to compare trajectory generators under common settings, addressing fragmentation and enabling reproducible assessment of multi-objective quality trade-offs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"505221f82f6c6c3ab5495c464edfebf9537e7809068553675b16f3fe615a62cd"},"motivation":"Urban trajectory generation is a fundamental task for transportation simulation, urban planning, and mobility analytics.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02287","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_62c91317432638c2","familyId":"catalog_family_62c91317432638c2","name":"CL-bench","oneLine":"CL-bench is an open-source benchmark with its own data and rubrics for evaluating models on coding and agentic tasks, scored using a setup fully aligned with the official procedure.","description":"CL-bench is an open-source benchmark with its own data and rubrics for evaluating models on coding and agentic tasks, scored using a setup fully aligned with the official procedure.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cl-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_62c91317432638c2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cl-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cl-bench","url":"https://llm-stats.com/benchmarks/cl-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_bcdfd03bd948d941","familyId":"catalog_family_bcdfd03bd948d941","name":"CL-bench (Life)","oneLine":"CL-bench Life variant.","description":"CL-bench Life variant.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cl-bench-(life)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bcdfd03bd948d941"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cl-bench-(life)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cl-bench-(life)","url":"https://llm-stats.com/benchmarks/cl-bench-(life)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_clarifycodebench_aa8d2946","familyId":"bmf_76530bd5505f","name":"ClarifyCodeBench","oneLine":"ClarifyCodeBench evaluates LLMs on clarifying ambiguous requirements for code generation through interactive dialogues. It includes manual annotations of ambiguity types, clarification questions, and ground-truth answers, with metrics for interaction quality.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation"],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.00711","pdf":"https://arxiv.org/pdf/2607.00711","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.00711"},"evidence":{"snippet":"To bridge this gap, we introduce ClarifyCodeBench, a novel interactive benchmark for evaluating LLMs' capability in resolving requirement ambiguity.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.00711"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ClarifyCodeBench evaluates LLMs on clarifying ambiguous requirements for code generation through interactive dialogues. It includes manual annotations of ambiguity types, clarification questions, and ground-truth answers, with metrics for interaction quality.","whyItMatters":"Code generation in practice involves underspecified requirements, but existing benchmarks assume perfect prompts. ClarifyCodeBench addresses this by measuring a critical yet underexplored capability, revealing that strong code generation does not imply effective requirement clarification.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"58ded43d2b3f7996b63a9725c4d6d88976d54d37f9a558b36f34b6a3ab737f7c"},"motivation":"Large Language Models have emerged as programming assistants.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.00711","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_classiclogic_43434cd2","familyId":"bmf_9984193cf2db","name":"ClassicLogic","oneLine":"ClassicLogic is a benchmark suite of four classic logic puzzles (Sudoku, KenKen, Kakuro, Futoshiki) with a hierarchical knowledge base that defines complex strategies as compositions of simpler ones. It evaluates an agent's compositional generalization in problem-solving from basic rules to multi-step strategies across increasing puzzle difficulty.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.05185","pdf":"https://arxiv.org/pdf/2607.05185","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.05185"},"evidence":{"snippet":"We introduce ClassicLogic, a new benchmark suite designed to evaluate an agent's ability to learn and compose problem-solving strategies.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.05185"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ClassicLogic is a benchmark suite of four classic logic puzzles (Sudoku, KenKen, Kakuro, Futoshiki) with a hierarchical knowledge base that defines complex strategies as compositions of simpler ones. It evaluates an agent's compositional generalization in problem-solving from basic rules to multi-step strategies across increasing puzzle difficulty.","whyItMatters":"Most compositional generalization benchmarks focus on language; ClassicLogic provides a structured, non-linguistic testbed with explicit compositional logic. It enables fine-grained assessment of reasoning capabilities and supports development of neuro-symbolic systems capable of systematic problem-solving.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fa6e7171734804034fc7bea25e3d4675f1b364ec427822776a6ac2bd3a51dbe1"},"motivation":"Compositional generalization, the ability to understand and produce novel combinations of known components, remains a fundamental challenge for modern artificial intelligence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.05185","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ClassicLogic Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.05185","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_cross-lingual-biography-enrichment-via-cla_8d142b84","familyId":"bmf_4ef127d105d2","name":"CLAW-4L","oneLine":"Evaluates cross-lingual biography enrichment on 300 Wikipedia biography pairs with claim annotations and a fine-grained claim-pair relation corpus.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":0.6,"links":{"report":"http://arxiv.org/abs/2608.23390v1","pdf":"https://arxiv.org/pdf/2608.23390v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Focusing on women from non-English-speaking contexts, we introduce \\textsc{CLAW-4L}, a benchmark consisting of 300 Wikipedia biography pairs linking an English biography with its French, Chinese or Azerbaijani counterpart, along with claim annotations and a fine-grained claim-pair relation corpus.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23390"},"ranking":{"today":{"score":50,"rank":6,"coverage":0.4,"confidence":"Low"},"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates cross-lingual biography enrichment on 300 Wikipedia biography pairs with claim annotations and a fine-grained claim-pair relation corpus.","whyItMatters":"Provides a resource for improving coverage of underrepresented biographies by leveraging non-English Wikipedia evidence.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"3708376119f01f3c20dd4c3fedc1afc09a24b69b0befaf1a0ffd272f3aca68ec"},"motivation":"English Wikipedia is often treated as the default encyclopedic source, yet non-English Wikipedia editions can contain richer locally grounded information for long-tail figures.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally introduced with annotated data and a clear task definition, and can be used by other researchers for claim extraction and alignment.","canonicalNameSource":"abstract","canonicalNameEvidence":"introduce \\textsc{CLAW-4L}, a benchmark consisting of 300 Wikipedia biography pairs"},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 main conference","evidence":"accepted by EMNLP 2026 main conference","evidenceUrl":"http://arxiv.org/abs/2608.23390v1","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-25T16:42:57.448509Z"},"venueAttempts":[{"venueName":"EMNLP 2026 main conference","reviewStatus":"accepted","decisionRaw":"accepted by EMNLP 2026 main conference","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"http://arxiv.org/abs/2608.23390v1","observedAt":"2026-08-25T16:42:57.448509Z","rawValue":"accepted by EMNLP 2026 main conference","level":"author-claim"}]}],"attentionForecast":{"score":48,"confidence":"Low","horizon":"7d","reason":"The benchmark targets cross-lingual knowledge enrichment with a modest dataset size and no immediate code or data link, limiting its likely attention in the first week."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_claw-anything_3c56b6b4","familyId":"bmf_ebd37017ecf1","name":"Claw-Anything","oneLine":"Claw-Anything evaluates always-on LLM personal assistants in simulated environments with long-horizon activity histories, interdependent backend services, and GUI/CLI across devices. It includes 200 human-verified tasks scored on completion, robustness, communication, and safety, with a live leaderboard.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26086","pdf":"https://arxiv.org/pdf/2605.26086","project":"https://libercoders.github.io/Claw-Anything/","code":"https://github.com/LiberCoders/Claw-Anything","data":"https://huggingface.co/datasets/LiberCoders/Claw-Anything","hfPaper":"https://huggingface.co/papers/2605.26086"},"evidence":{"snippet":"To address this gap, we introduce Claw-Anything, a benchmark that expands agent context along three dimensions: long-horizon activity histories, interdependent backend services, and integrated GUI and CLI interaction across multiple devices.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-reviewed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":25,"hfDailySubmittedAt":null,"githubStars":48,"githubScope":"benchmark_repo","hfDatasetDownloads":80,"hfDatasetLikes":10},"source":{"type":"arxiv","id":"2605.26086"},"ranking":{},"description":"Claw-Anything evaluates always-on LLM personal assistants in simulated environments with long-horizon activity histories, interdependent backend services, and GUI/CLI across devices. It includes 200 human-verified tasks scored on completion, robustness, communication, and safety, with a live leaderboard.","whyItMatters":"It expands agent evaluation to broad, always-on contexts, revealing capability gaps in stateful, proactive assistance and supporting scalable data generation for training.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a4ba1f383be74e2a945572f6c15d0cad22707c80dfa7a662d4ae181581c0c443"},"motivation":"Large language model agents are increasingly envisioned as always-on personal assistants with access to anything relevant in the user's digital world.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"source-reviewed","reviewedAt":"2026-08-19","sources":["https://github.com/LiberCoders/Claw-Anything","https://huggingface.co/datasets/LiberCoders/Claw-Anything","https://arxiv.org/abs/2605.26086"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26086","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"releaseDates":{"firstPublicAt":"2026-05-25","paperV1At":"2026-05-25"},"publishers":[{"name":"LiberCoders","organizationType":"community","sourceUrl":"https://github.com/LiberCoders/Claw-Anything","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_837a019e46144f8d","familyId":"catalog_family_837a019e46144f8d","name":"Claw-Eval","oneLine":"Claw-Eval tests real-world agentic task completion across complex multi-step scenarios, evaluating a model's ability to use tools, navigate environments, and complete end-to-end tasks autonomously.","description":"Claw-Eval tests real-world agentic task completion across complex multi-step scenarios, evaluating a model's ability to use tools, navigate environments, and complete end-to-end tasks autonomously.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2604.06132","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_837a019e46144f8d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/claw-eval"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/claw-eval"}],"catalogSources":[{"catalog":"benchlm","sourceId":"clawEval","url":"https://benchlm.ai/benchmarks/claw-eval","paperUrl":"https://arxiv.org/abs/2604.06132","year":"2026","fullName":"Claw-Eval","format":"End-to-end autonomous-agent evaluation with Pass^3 scoring","tasks":"300 tasks, 2,159 rubrics","successorKey":null},{"catalog":"llm-stats","sourceId":"claw-eval","url":"https://llm-stats.com/benchmarks/claw-eval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","agents","code"],"catalogModelCount":14,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_claw-swe-bench_ea69cc9d","familyId":"bmf_32be4dea95f8","name":"Claw-SWE-Bench","oneLine":"Claw-SWE-Bench is a multilingual SWE-bench-style benchmark and adapter protocol for comparing agent harnesses on coding tasks, with 350 GitHub issue-resolution instances across 8 languages and 43 repositories. Score is Pass@1 on patch correctness.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.12344","pdf":"https://arxiv.org/pdf/2606.12344","project":null,"code":"https://github.com/opensquilla/claw-swe-bench","data":"https://huggingface.co/datasets/TokenRhythm/Claw-SWE-Bench","hfPaper":"https://huggingface.co/papers/2606.12344"},"evidence":{"snippet":"We introduce Claw-SWE-Bench, a multilingual SWE-bench-style benchmark and adapter protocol that makes heterogeneous agent harnesses, or claws, comparable under fair settings including a fixed prompt, runtime budget, workspace contract, patch extraction procedure, and evaluator.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":71,"hfDailySubmittedAt":"2026-06-11T00:00:00.000Z","githubStars":97,"githubScope":"benchmark_repo","hfDatasetDownloads":1658,"hfDatasetLikes":6},"source":{"type":"arxiv","id":"2606.12344"},"ranking":{"90d":{"score":64,"rank":13,"coverage":1.0,"confidence":"High","datasetDownloadRank":11,"datasetRankPopulation":66}},"description":"Claw-SWE-Bench is a multilingual SWE-bench-style benchmark and adapter protocol for comparing agent harnesses on coding tasks, with 350 GitHub issue-resolution instances across 8 languages and 43 repositories. Score is Pass@1 on patch correctness.","whyItMatters":"Enables fair comparison of autonomous coding agents by treating harness and cost as first-class evaluation axes. Useful for developers and researchers building general-purpose coding agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f8bb0766fa75a120dfa40c1479579a5d65c458ccac51b1b83bcc162300dbf3fb"},"motivation":"General-purpose agents such as OpenClaw are increasingly used as autonomous tool users, but their coding ability is difficult to measure under SWE-bench: a generic agent does not by itself satisfy the clean Docker workspace, patch, and prediction contract required for scoring.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.12344","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TokenRhythm Technologies","organizationType":"company-research-lab","sourceUrl":"https://github.com/opensquilla/claw-swe-bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_clawarena-team_8c93773c","familyId":"bmf_c7eb98430ee5","name":"ClawArena-Team","oneLine":"ClawArena-Team evaluates a single text-only LLM's ability to manage a fixed, locally served pool of subagents (LLM, VLM, omni) across 41 multi-turn, multimodal, multi-directory scenarios with 258 evaluation rounds and 72 staged updates. Scoring is execution-based via shell commands, producing a Subagent-Management Score (SMS) that multiplies task correctness by a least-privilege and modality-routing factor, without LLM judges.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.31174","pdf":"https://arxiv.org/pdf/2606.31174","project":"https://www.clawarena.cc/","code":"https://github.com/aiming-lab/ClawArena","data":null,"hfPaper":"https://huggingface.co/papers/2606.31174"},"evidence":{"snippet":"We introduce ClawArena-Team, a benchmark of 41 multi-turn, multimodal, multi-directory scenarios spanning 258 evaluation rounds and 72 staged updates that measures this management ability.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":63,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31174"},"ranking":{"90d":{"score":46,"rank":95,"coverage":0.7,"confidence":"Medium"}},"description":"ClawArena-Team evaluates a single text-only LLM's ability to manage a fixed, locally served pool of subagents (LLM, VLM, omni) across 41 multi-turn, multimodal, multi-directory scenarios with 258 evaluation rounds and 72 staged updates. Scoring is execution-based via shell commands, producing a Subagent-Management Score (SMS) that multiplies task correctness by a least-privilege and modality-routing factor, without LLM judges.","whyItMatters":"Existing agent benchmarks measure a policy's own task-solving or emergent behavior of fixed multi-agent systems, but not the leadership capability of a single model orchestrating subagents. ClawArena-Team fills this gap by isolating management skill from raw capability, supporting decisions on model selection for delegation-heavy workflows and revealing cost-quality trade-offs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9666a2ac4bc96807405ca733dd9d6cc3d5253f3cac5e46e4dde649dd478f78fc"},"motivation":"Production large language-model (LLM) agents are increasingly deployed not as lone problem-solvers but as managers: a main model creates specialized subagents, delegates work, and orchestrates their parallel, asynchronous returns through dynamic workflows.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31174","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_2c9490b45364dd97","familyId":"catalog_family_2c9490b45364dd97","name":"ClawEval-MM","oneLine":"ClawEval-MM is the multimodal variant of ClawEval, evaluating agentic problem solving with visual inputs.","description":"ClawEval-MM is the multimodal variant of ClawEval, evaluating agentic problem solving with visual inputs.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Agents","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/claw-eval-mm","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2c9490b45364dd97"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/claw-eval-mm"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"claw-eval-mm","url":"https://llm-stats.com/benchmarks/claw-eval-mm","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","agents","vision"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_clawprobench_7f331d78","familyId":"bmf_dbc550709e63","name":"ClawProBench","oneLine":"Evaluates agent configurations on 102 live-runtime and 68 frozen holdout scenarios, scoring execution traces with a safety-gated formula covering correctness, process quality, and efficiency.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.22510v1","pdf":"https://arxiv.org/pdf/2608.22510v1","project":null,"code":null,"data":"https://huggingface.co/datasets/xyh110sym/clawprobench","hfPaper":null},"evidence":{"snippet":"We present ClawProBench, a trace-aware benchmark for runtime-native agent evaluation instantiated on OpenClaw, a live agent runtime with workspace tools and native surfaces for browsing, memory, messaging, scheduling, skills, and subagents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":2,"hfDailySubmittedAt":"2026-08-25T00:00:00.000Z","githubStars":822,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22510"},"ranking":{"30d":{"score":76,"rank":17,"coverage":0.85,"confidence":"High"},"90d":{"score":78,"rank":26,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates agent configurations on 102 live-runtime and 68 frozen holdout scenarios, scoring execution traces with a safety-gated formula covering correctness, process quality, and efficiency.","whyItMatters":"It moves agent evaluation beyond final answers to expose runtime-navigation, safety-boundary, and repeated-execution failures that conventional leaderboards hide.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"ccfe8c9bfb02e81e45230688dc1230abfdfede0e45b8b2324e6d6c829b3b0fd0"},"motivation":"Agent benchmarks often evaluate only final answers even when agents run on stateful runtimes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"source-reviewed","reviewedAt":"2026-08-30","model":"deepseek-v4-pro","decisionReason":"The paper defines two scored tracks, safety-gated scoring, and evaluation manifests with sanitized traces, providing a credible comparison path.","canonicalNameSource":"abstract","canonicalNameEvidence":"We present ClawProBench, a trace-aware benchmark for runtime-native agent evaluation instantiated on OpenClaw","sources":["https://arxiv.org/abs/2608.22510","https://github.com/suyoumo/ClawProBench","https://huggingface.co/datasets/xyh110sym/clawprobench"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22510v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":72,"confidence":"Medium","horizon":"7d","reason":"A trace-aware agent benchmark with a frozen holdout and multi-configuration results addresses a current gap and should attract research community attention."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_clawtrack_7bde407b","familyId":"bmf_486094563286","name":"ClawTrack","oneLine":"ClawTrack is a dual-assessment benchmark for agents, measuring task outcomes and process quality across 320 tasks in 8 domains with 25+ mock services.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.28037","pdf":"https://arxiv.org/pdf/2607.28037","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.28037"},"evidence":{"snippet":"In this work, we present ClawTrack, a dual-assessment benchmark that simultaneously measures what an agent achieves (Task Score) and how it achieves it (Process Score).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.28037"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"ClawTrack is a dual-assessment benchmark for agents, measuring task outcomes and process quality across 320 tasks in 8 domains with 25+ mock services.","whyItMatters":"It aims to decompose agent success into reasoning dimensions, which could improve attribution and post-training filtering.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2a8741ac7160ee81fa9e4468214f463d5efa3b45ef56d79b3b01a14e5b460470"},"motivation":"As LLM-based agents are deployed in complex, multi-step workflows, a critical evaluation gap has emerged: most existing benchmarks judge only final outcomes, unable to distinguish reliable reasoning from lucky success or attribute failures to specific process deficiencies, hindering attribution in long-horizon tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.28037","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_clbench-v_d435f082","familyId":"bmf_b6a0b47faa26","name":"CLBench-V","oneLine":"CLBench-V evaluates multimodal context learning across three dimensions: context grounding, new information application, and new knowledge learning. It includes 3,443 instances across 14 subdatasets spanning science, finance, long-document understanding, spatial reasoning, and web-based VQA.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25294","pdf":"https://arxiv.org/pdf/2607.25294","project":null,"code":"https://github.com/IamLihua/CLBench-V","data":null,"hfPaper":"https://huggingface.co/papers/2607.25294"},"evidence":{"snippet":"We introduce CLBench-V, a benchmark for multimodal context learning that addresses the difficulty of localizing where context use breaks down by organizing tasks around three dimensions: context grounding, new information application, and new knowledge learning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":49,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25294"},"ranking":{"90d":{"score":36,"rank":189,"coverage":0.7,"confidence":"Medium"}},"description":"CLBench-V evaluates multimodal context learning across three dimensions: context grounding, new information application, and new knowledge learning. It includes 3,443 instances across 14 subdatasets spanning science, finance, long-document understanding, spatial reasoning, and web-based VQA.","whyItMatters":"Existing context learning benchmarks focus on text, missing multimodal settings where context is in figures, tables, and maps. CLBench-V provides a structured evaluation to localize where context use breaks down, aiding progress in multimodal models for real-world tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"43a1a78b46d23c2eb8796c5c81c281a3cfd93d5391e8d028810454aea4ef2645"},"motivation":"Real-world tasks often require models to learn from task-specific context rather than relying only on pre-trained knowledge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25294","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"CLBench-V Team","organizationType":"academic-lab","sourceUrl":"https://github.com/IamLihua/CLBench-V","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_e5c7ffac26fed654","familyId":"catalog_family_e5c7ffac26fed654","name":"CLIcK","oneLine":"Evaluates Korean culture and linguistics.","description":"Evaluates Korean culture and linguistics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Korean"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://benchlm.ai/benchmarks/click","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e5c7ffac26fed654"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/click"}],"catalogSources":[{"catalog":"benchlm","sourceId":"click","url":"https://benchlm.ai/benchmarks/click","paperUrl":null,"year":null,"fullName":"Cultural and Linguistic Intelligence in Korean","format":"Cultural/linguistic QA","tasks":"1,995 questions","successorKey":null}],"catalogCategories":["korean"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_clinenv_3008fd36","familyId":"bmf_e93c43eae3f3","name":"ClinEnv","oneLine":"ClinEnv evaluates LLMs as physicians in an interactive multi-stage EHR simulation, requiring queries to four specialized agents before committing to medical decisions, scored via ontology-grounded matching.","area":"Agents & Tool Use","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02568","pdf":"https://arxiv.org/pdf/2606.02568","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02568"},"evidence":{"snippet":"We present ClinEnv, an interactive benchmark that evaluates LLMs as attending physicians over real inpatient admissions under a paradigm we term Longitudinal Inpatient Simulation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02568"},"ranking":{},"description":"ClinEnv evaluates LLMs as physicians in an interactive multi-stage EHR simulation, requiring queries to four specialized agents before committing to medical decisions, scored via ontology-grounded matching.","whyItMatters":"Measures both decision quality and information-gathering process, exposing a gap between them that outcome-only evaluation misses, which is critical for clinical decision support.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"773ede85e4bd39a070cd8c88304b2a93444095206df712d0d4d767abb48f0123"},"motivation":"Clinical practice is not the selection of an answer from enumerated options: a physician gathers heterogeneous information incrementally and commits to sequential, irreversible decisions under uncertainty.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02568","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_clinhallu_85f948a0","familyId":"bmf_52fa2852cf5f","name":"ClinHallu","oneLine":"ClinHallu is a benchmark for diagnosing stage-wise hallucinations in medical multimodal large language models. It contains 7,031 instances with structured reasoning traces decomposed into visual recognition, knowledge recall, and reasoning integration, along with stage-replacement interventions for measuring final answer changes.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.14697","pdf":"https://arxiv.org/pdf/2606.14697","project":null,"code":"https://github.com/alibaba-damo-academy/ClinHallu","data":null,"hfPaper":"https://huggingface.co/papers/2606.14697"},"evidence":{"snippet":"To enable source-level hallucination diagnosis, we introduce ClinHallu, a benchmark for stage-wise hallucination diagnosis in medical MLLM reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":"2026-06-15T00:00:00.000Z","githubStars":9,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.14697"},"ranking":{"90d":{"score":38,"rank":165,"coverage":0.7,"confidence":"Medium"}},"description":"ClinHallu is a benchmark for diagnosing stage-wise hallucinations in medical multimodal large language models. It contains 7,031 instances with structured reasoning traces decomposed into visual recognition, knowledge recall, and reasoning integration, along with stage-replacement interventions for measuring final answer changes.","whyItMatters":"Existing medical hallucination benchmarks often ignore the source of hallucinations within reasoning. ClinHallu allows for fine-grained diagnosis of where errors originate, providing a testbed for improving model reliability in clinical decision support.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c8e07fb367894c41ba360c6edec112e26ee266686dcf9cb604385753e9d977d2"},"motivation":"Building trustworthy medical multimodal large language models (MLLMs) is critical for reliable clinical decision support.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.14697","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Alibaba DAMO Academy","organizationType":"company-research-lab","sourceUrl":"https://github.com/alibaba-damo-academy/ClinHallu","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_clinicalmc_22427e9b","familyId":"bmf_297bec9ba523","name":"ClinicalMC","oneLine":"ClinicalMC evaluates LLM clinical decision-making across multi-course patient trajectories, with 1,275 Chinese and 5,804 English samples spanning four stages from admission to discharge. Includes triage, examination, diagnosis, treatment, and final diagnosis.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03157","pdf":"https://arxiv.org/pdf/2606.03157","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03157"},"evidence":{"snippet":"To address this gap, we propose ClinicalMC, a benchmark for multi-course clinical decision-making.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03157"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ClinicalMC evaluates LLM clinical decision-making across multi-course patient trajectories, with 1,275 Chinese and 5,804 English samples spanning four stages from admission to discharge. Includes triage, examination, diagnosis, treatment, and final diagnosis.","whyItMatters":"Existing clinical benchmarks focus on single-course scenarios, missing the complexity of evolving patient conditions. ClinicalMC enables evaluation of dynamic multi-turn decision-making, supporting safer LLM deployment in healthcare.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6376e4263cd7f0f2ac1d5cfd8a45f693e5bf8729c2a91823ffa73bfd38025e94"},"motivation":"Large language models (LLMs) have been widely adopted in healthcare, yet they still encounter significant challenges in complex clinical decision-making scenarios.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03157","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_clinicare-bench_6faa931e","familyId":"bmf_23f50f44501d","name":"CliniCARE-Bench","oneLine":"CliniCARE-Bench evaluates clinical agents on retrospective audit tasks over longitudinal EHR data, measuring verdict accuracy, evidence grounding, process adherence, calibrated abstention, reliability, and efficiency.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07796","pdf":"https://arxiv.org/pdf/2608.07796","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07796"},"evidence":{"snippet":"We introduce CliniCARE-Bench (Clinical Calibrated Audit of Medical Reasoning in EHR), a benchmark for retrospective clinical audit: 25 clinician-validated scenarios instantiated as 750 patient-specific cases over real-patient-derived MIMIC-IV data.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07796"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CliniCARE-Bench evaluates clinical agents on retrospective audit tasks over longitudinal EHR data, measuring verdict accuracy, evidence grounding, process adherence, calibrated abstention, reliability, and efficiency.","whyItMatters":"Provides a deployment-oriented benchmark for clinical agents, assessing not just accuracy but also defensibility and calibration, which are essential for trustworthy clinical decision support.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"71096404cea83d695392b31e689eaf23aa9d62f96a486bcbb0535b76d507d261"},"motivation":"Large language models perform strongly on medical knowledge benchmarks, but reliable clinical deployment requires agents to conduct defensible investigations over heterogeneous, longitudinal records: determining what evidence is needed, retrieving and reconciling structured and free-text data, grounding conclusions in verifiable evidence, and deferring cases that cannot be resolved reliably.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07796","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_clinlens_3ca0c7ce","familyId":"bmf_0887f3ebbae5","name":"ClinLens","oneLine":"ClinLens evaluates clinical data-science agents on 200 executable tasks over five linked MIMIC resources (EHR, notes, ECG, chest X-rays, echocardiograms), organized by a 4x5 taxonomy of patient-time scopes and analysis capabilities. Scoring uses a STRICTPASS metric requiring correct artifacts, semantics, and final answers.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.26155","pdf":"https://arxiv.org/pdf/2607.26155","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.26155"},"evidence":{"snippet":"We introduce CLINLENS, a benchmark of 200 executable tasks over five linked MIMIC resources spanning structured electronic health records, notes, electrocardiograms, chest radiographs, and echocardiograms.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26155"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ClinLens evaluates clinical data-science agents on 200 executable tasks over five linked MIMIC resources (EHR, notes, ECG, chest X-rays, echocardiograms), organized by a 4x5 taxonomy of patient-time scopes and analysis capabilities. Scoring uses a STRICTPASS metric requiring correct artifacts, semantics, and final answers.","whyItMatters":"Existing benchmarks isolate medical QA or table reasoning, lacking integrated longitudinal clinical data science. ClinLens fills this gap with a program-first reverse synthesis approach, exposing a gap between runnable code and correct clinical analysis, guiding improvements in clinical agent reliability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2514840c4290af7724ea0fced91dd766511f6a845a9ed87a6edd46736e690d6f"},"motivation":"Clinical data-science agents must transform heterogeneous longitudinal records into auditable analyses, yet existing benchmarks largely isolate medical question answering, structured-table reasoning, or generic scientific repositories.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26155","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_clinocr-bench_8be5c7a7","familyId":"bmf_6c9fc7398ebb","name":"ClinOCR-Bench","oneLine":"ClinOCR-Bench is a public dataset of 384 scanned clinical documents across six artifact subsets (normal, handwriting, poor quality, rotation, tables, mixed). It evaluates OCR systems on clinical text extraction, with ground truth transcriptions and template-aware train/test splits supporting 0-shot and 1-shot evaluation.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.03650","pdf":"https://arxiv.org/pdf/2607.03650","project":null,"code":"https://github.com/ClinOCR-Bench/ClinOCR-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.03650"},"evidence":{"snippet":"Therefore, we release a publicly available, realistic-looking OCR benchmark dataset, ClinOCR-Bench, with 384 scanned images across 6 subsets: Normal, Handwriting, Poor Quality, Rotation, Tables, and Mix-artifacts.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.03650"},"ranking":{"90d":{"score":33,"rank":219,"coverage":0.55,"confidence":"Low"}},"description":"ClinOCR-Bench is a public dataset of 384 scanned clinical documents across six artifact subsets (normal, handwriting, poor quality, rotation, tables, mixed). It evaluates OCR systems on clinical text extraction, with ground truth transcriptions and template-aware train/test splits supporting 0-shot and 1-shot evaluation.","whyItMatters":"Existing clinical OCR evaluations rely on private data lacking common scan artifacts. ClinOCR-Bench provides a standardized, realistic set of clinical documents with controlled artifacts, enabling reproducible comparison of OCR and vision-language models on real-world clinical scanning challenges.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7fa6bc0c3ec964067f111d5cdd8036b541d6ebdf4c991b8b6b834230ce37a648"},"motivation":"Extracting textual information from scanned medical documents, such as external laboratory reports and manually filled forms, has been a major challenge in modern electronic health records (EHRs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.03650","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_clip-cc-bench_6e2be6c1","familyId":"bmf_250f7a769d65","name":"CLIP-CC-Bench","oneLine":"CLIP-CC-Bench evaluates paragraph-level video description quality using 5-hour movie content and expert-written references. It scores 17 VLMs via coarse- and fine-grained semantic matching with five LLM-based embedding judges.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.04302","pdf":"https://arxiv.org/pdf/2608.04302","project":null,"code":"https://github.com/Multimodal-Intelligence-Lab/CLIP-CC-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.04302"},"evidence":{"snippet":"We introduce CLIP-CC-Bench, an evaluation suite for long-form video description built from 5 hours of movie content segmented into 90-second clips, each paired with an expert-written paragraph-style reference.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":"2026-08-10T00:00:00.000Z","githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04302"},"ranking":{"30d":{"score":19,"rank":161,"coverage":0.85,"confidence":"High"},"90d":{"score":23,"rank":309,"coverage":0.7,"confidence":"Medium"}},"description":"CLIP-CC-Bench evaluates paragraph-level video description quality using 5-hour movie content and expert-written references. It scores 17 VLMs via coarse- and fine-grained semantic matching with five LLM-based embedding judges.","whyItMatters":"It fills the gap in long-form video description benchmarks, providing a reliable framework with public scripts and data for reproducible evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"eab2cf54c1d7768789183433e46ea8b124eb03232526c372d3be4133bcf43833"},"motivation":"Benchmarking video-language models has largely focused on short clips and single-sentence metrics, leaving open whether current systems can generate accurate long-form, paragraph-level descriptions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"and presented at EvalMG 2026, the Second Workshop on Evaluation for Multimodal Generation, co-located with ACM SIGIR 202","evidence":"Accepted and presented at EvalMG 2026, the Second Workshop on Evaluation for Multimodal Generation, co-located with ACM SIGIR 2026","evidenceUrl":"https://arxiv.org/abs/2608.04302","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"and presented at EvalMG 2026, the Second Workshop on Evaluation for Multimodal Generation, co-located with ACM SIGIR 202","reviewStatus":"accepted","decisionRaw":"Accepted and presented at EvalMG 2026, the Second Workshop on Evaluation for Multimodal Generation, co-located with ACM SIGIR 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.04302","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted and presented at EvalMG 2026, the Second Workshop on Evaluation for Multimodal Generation, co-located with ACM SIGIR 2026","level":"author-claim"}]}],"publishers":[{"name":"Multimodal Intelligence Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/Multimodal-Intelligence-Lab/CLIP-CC-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_clir-bench_e1c31f75","familyId":"bmf_eab8b70b905c","name":"CLIR-Bench","oneLine":"A benchmark for question answering over irregular clinical time series from ICU records, containing 6,600 QA instances across 11 clinical variables and 11 tasks. It evaluates answer accuracy and evidence use through explicit temporal evidence and answer derivation rules.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.09880","pdf":"https://arxiv.org/pdf/2607.09880","project":null,"code":null,"data":"https://huggingface.co/datasets/winall/CLIR-Bench","hfPaper":"https://huggingface.co/papers/2607.09880"},"evidence":{"snippet":"To fill this gap, we introduce CLIR-Bench, a benchmark for irregular clinical time series QA constructed from de-identified ICU records through a principled four-stage pipeline.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":59,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2607.09880"},"ranking":{"90d":{"score":38,"rank":166,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":58,"datasetRankPopulation":66}},"description":"A benchmark for question answering over irregular clinical time series from ICU records, containing 6,600 QA instances across 11 clinical variables and 11 tasks. It evaluates answer accuracy and evidence use through explicit temporal evidence and answer derivation rules.","whyItMatters":"Existing benchmarks focus on regular time-series or static medical QA, while real ICU data is sparse and asynchronous. This benchmark provides a way to assess whether models can reason over irregular temporal evidence, addressing a gap in clinical NLP evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5dbfca2cc4764cc24ea4cbcdeb5f3380d85699a013877f6ee49c5f0213f3e6af"},"motivation":"Clinical time series are central to patient monitoring, risk assessment, and clinical decision support.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.09880","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"CLIR-Bench Team","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/datasets/winall/CLIR-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_3102d317f87189a6","familyId":"catalog_family_3102d317f87189a6","name":"CloningScenarios","oneLine":"CloningScenarios is an expert-level multi-step reasoning benchmark about difficult genetic cloning scenarios in multiple-choice format. It evaluates dual-use biological knowledge relevant to bioweapons development.","description":"CloningScenarios is an expert-level multi-step reasoning benchmark about difficult genetic cloning scenarios in multiple-choice format. It evaluates dual-use biological knowledge relevant to bioweapons development.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Reasoning","Safety","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cloningscenarios","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3102d317f87189a6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cloningscenarios"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cloningscenarios","url":"https://llm-stats.com/benchmarks/cloningscenarios","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","safety","healthcare"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_closer-bench_7847c9e5","familyId":"bmf_01919ee4d7a3","name":"CLOSER-Bench","oneLine":"CLOSER-Bench evaluates budgeted cross-stage design closure for hardware agents with spec-to-RTL, RTL-to-GDS, and spec-to-GDS tasks, using open-source tools and recording quality, progress, tool cost, and recovery.","area":"Code & Software","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Semiconductors","Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.16632","pdf":"https://arxiv.org/pdf/2607.16632","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.16632"},"evidence":{"snippet":"We introduce CLOSER-Bench, a controlled evaluation protocol for budgeted cross-stage design closure.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.16632"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CLOSER-Bench evaluates budgeted cross-stage design closure for hardware agents with spec-to-RTL, RTL-to-GDS, and spec-to-GDS tasks, using open-source tools and recording quality, progress, tool cost, and recovery.","whyItMatters":"Provides a controlled protocol for hardware design closure, addressing the gap in evaluating agents across abstraction boundaries.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"acbab26b48fbb48e1c0fbb3cd0dbdb9c9c75271206a0fe522f03dc30e0d2dd18"},"motivation":"Hardware engineering exposes coding agents to a form of long-horizon work that is difficult to capture with pass-at-k: progress is continuous, tool feedback is delayed and heterogeneous, and a backend failure may require revising RTL rather than tuning another physical-design parameter.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.16632","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"The authors","organizationType":"community","sourceUrl":"https://arxiv.org/abs/2607.16632","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"specific"},{"id":"bm_cloudcons_9f1d2495","familyId":"bmf_7839980df73c","name":"CloudCons","oneLine":"CloudCons is an end-to-end benchmark for evaluating forecasting models in cloud resource consolidation. It encompasses datasets from Huawei Cloud, Microsoft Azure, and Google Borg with diverse workload patterns, and evaluates statistical, deep learning, and time series foundation models. The evaluation includes resource efficiency and service reliability metrics under a forecast-then-optimize paradigm.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13513","pdf":"https://arxiv.org/pdf/2606.13513","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.13513"},"evidence":{"snippet":"To bridge this gap, we propose CloudCons, a comprehensive end-to-end benchmark designed to evaluate forecasting models within the specific context of cloud resource consolidation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13513"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CloudCons is an end-to-end benchmark for evaluating forecasting models in cloud resource consolidation. It encompasses datasets from Huawei Cloud, Microsoft Azure, and Google Borg with diverse workload patterns, and evaluates statistical, deep learning, and time series foundation models. The evaluation includes resource efficiency and service reliability metrics under a forecast-then-optimize paradigm.","whyItMatters":"Existing benchmarks focus only on prediction error, leaving the downstream decision utility of forecasting models unverified. CloudCons addresses this gap by assessing practical value in cloud resource consolidation, offering insights into balancing resource efficiency and service reliability, aiding deployment decisions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cef34bc8597b5f95de0aff24a0814fcb91b64cd035c81390cd5da7407290cd36"},"motivation":"Driven by conservative over-provisioning to guarantee service reliability, resource utilization in cloud data centers remains at low levels.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"KDD 2026","evidence":"Accepted to KDD 2026","evidenceUrl":"https://arxiv.org/abs/2606.13513","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"KDD 2026","reviewStatus":"accepted","decisionRaw":"Accepted to KDD 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.13513","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to KDD 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_clubench_c6a13ff8","familyId":"bmf_22a6f1914a19","name":"CLUBench","oneLine":"CLUBench evaluates clustering algorithms across 131 datasets (tabular, text, image) using 24 algorithms, measuring clustering performance via standard metrics like NMI, ARI, etc.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29933","pdf":"https://arxiv.org/pdf/2605.29933","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29933"},"evidence":{"snippet":"To address this gap, we introduce CLUBench, a comprehensive clustering benchmark comprising 24 algorithms of diverse principles evaluated on 131 datasets across tabular, text, and image data, involving 178,815 experiments.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29933"},"ranking":{},"description":"CLUBench evaluates clustering algorithms across 131 datasets (tabular, text, image) using 24 algorithms, measuring clustering performance via standard metrics like NMI, ARI, etc.","whyItMatters":"There is no comprehensive comparison of classical, deep, and foundation-model clustering methods. CLUBench provides systematic insights into algorithm selection and performance across data types.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"158ee8839053bf817bcdc58d04d9f4faab1c3ba40bea0ab4db54de0ed202b987"},"motivation":"Clustering is a fundamental problem in data science with a long-standing research history, yielding numerous insightful algorithms.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29933","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_clueaegis-bench_30a26f44","familyId":"bmf_f604d3adcb02","name":"ClueAegis-Bench","oneLine":"A benchmark decomposing synthetic image detection into annotated forensic cognitive skills, used to evaluate a proposed detection framework.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-05-24","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.25009","pdf":"https://arxiv.org/pdf/2605.25009","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.25009"},"evidence":{"snippet":"To support this paradigm, we introduce \\textbf{ClueAegis-Bench}, which decomposes synthetic image detection into explicitly annotated forensic cognitive skills for structured evaluation beyond binary classification.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25009"},"ranking":{},"description":"A benchmark decomposing synthetic image detection into annotated forensic cognitive skills, used to evaluate a proposed detection framework.","whyItMatters":"Supports evidence-based synthetic image detection beyond binary classification.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e1a9944fc2dbaee73847881d165a1755c29f70fe5f78b0452841e4d961ddc873"},"motivation":"The rapid advancement of generative models has made synthetic images increasingly realistic, challenging reliable detection.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25009","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_f44ad3b35d2776f3","familyId":"catalog_family_f44ad3b35d2776f3","name":"CLUEWSC","oneLine":"CLUEWSC2020 is the Chinese version of the Winograd Schema Challenge, part of the CLUE benchmark. It focuses on pronoun disambiguation and coreference resolution, requiring models to determine which noun a pronoun refers to in a sentence. The dataset contains 1,244 training samples and 304 development samples extracted from contemporary Chinese literature.","description":"CLUEWSC2020 is the Chinese version of the Winograd Schema Challenge, part of the CLUE benchmark. It focuses on pronoun disambiguation and coreference resolution, requiring models to determine which noun a pronoun refers to in a sentence. The dataset contains 1,244 training samples and 304 development samples extracted from contemporary Chinese literature.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Language"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f44ad3b35d2776f3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/cluewsc"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cluewsc"}],"catalogSources":[{"catalog":"benchlm","sourceId":"cluewsc","url":"https://benchlm.ai/benchmarks/cluewsc","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"CLUEWSC","format":"Exact match","tasks":"Chinese coreference questions","successorKey":null},{"catalog":"llm-stats","sourceId":"cluewsc","url":"https://llm-stats.com/benchmarks/cluewsc","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","language"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d7703d0042f402be","familyId":"catalog_family_d7703d0042f402be","name":"CMath","oneLine":"A Chinese mathematical reasoning benchmark reported in DeepSeek-V4 base-model evaluations.","description":"A Chinese mathematical reasoning benchmark reported in DeepSeek-V4 base-model evaluations.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d7703d0042f402be"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/cmath"}],"catalogSources":[{"catalog":"benchlm","sourceId":"cmath","url":"https://benchlm.ai/benchmarks/cmath","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"CMath","format":"Exact match","tasks":"Chinese math problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_5694f82dafbace0e","familyId":"catalog_family_5694f82dafbace0e","name":"CMMLU","oneLine":"CMMLU (Chinese Massive Multitask Language Understanding) is a comprehensive Chinese benchmark that evaluates the knowledge and reasoning capabilities of large language models across 67 different subject topics. The benchmark covers natural sciences, social sciences, engineering, and humanities with multiple-choice questions ranging from basic to advanced professional levels.","description":"CMMLU (Chinese Massive Multitask Language Understanding) is a comprehensive Chinese benchmark that evaluates the knowledge and reasoning capabilities of large language models across 67 different subject topics. The benchmark covers natural sciences, social sciences, engineering, and humanities with multiple-choice questions ranging from basic to advanced professional levels.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Language","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5694f82dafbace0e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/cmmlu"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cmmlu"}],"catalogSources":[{"catalog":"benchlm","sourceId":"cmmlu","url":"https://benchlm.ai/benchmarks/cmmlu","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"Chinese Massive Multitask Language Understanding","format":"Exact match","tasks":"Chinese academic QA","successorKey":null},{"catalog":"llm-stats","sourceId":"cmmlu","url":"https://llm-stats.com/benchmarks/cmmlu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","language","reasoning","general"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_ba5232cc57812540","familyId":"catalog_family_ba5232cc57812540","name":"CMT-Benchmark","oneLine":"CMT-Benchmark evaluates models on condensed matter theory problems, testing advanced physics reasoning across areas such as many-body systems, quantum field theory, and statistical mechanics.","description":"CMT-Benchmark evaluates models on condensed matter theory problems, testing advanced physics reasoning across areas such as many-body systems, quantum field theory, and statistical mechanics.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Physics","Reasoning","Science"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cmt-benchmark","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ba5232cc57812540"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cmt-benchmark"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cmt-benchmark","url":"https://llm-stats.com/benchmarks/cmt-benchmark","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["physics","reasoning","science"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_cn-newstts-bench_e74cef36","familyId":"bmf_c1b15a4e7ff6","name":"CN-NewsTTS Bench","oneLine":"CN-NewsTTS Bench v0.1 evaluates Chinese news TTS pronunciation of dense written forms (scores, model names, ranges, units, percentages, abbreviations) from raw text, using 800 public test records and a 992-target auto-evaluable subset with ASR-ensemble transcripts and an automatic target scorer.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24714","pdf":"https://arxiv.org/pdf/2606.24714","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.24714"},"evidence":{"snippet":"We introduce CN-NewsTTS Bench v0.1, an open target-level benchmark for evaluating whether Chinese news TTS products pronounce such targets correctly from raw text, without user-side rules, LLM rewriting, SSML hints, or manual edits.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24714"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CN-NewsTTS Bench v0.1 evaluates Chinese news TTS pronunciation of dense written forms (scores, model names, ranges, units, percentages, abbreviations) from raw text, using 800 public test records and a 992-target auto-evaluable subset with ASR-ensemble transcripts and an automatic target scorer.","whyItMatters":"Addresses the gap where TTS systems may preserve written strings while altering spoken meaning in Chinese news. Provides a reproducible target-level scoring contract for comparing pronunciation accuracy across systems without manual intervention.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"38c0273a14823ec91b9d167248829d6271e3703be38dc2f02ba287f2596a65e2"},"motivation":"Chinese news text contains dense written forms such as scores, hyphenated model names, ranges, unit symbols, percentages, English abbreviations, and mixed Chinese-Latin-digit names.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24714","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_cneo-bench_bbdf3de1","familyId":"bmf_f2e97f1a4ddb","name":"CNeo-Bench","oneLine":"Chinese neologisms exploit diverse and unique linguistic mechanisms, such as phonetic substitution (e.g., 886 for ``bye-bye'') and visual character decomposition that are rare in other languages.","area":"Language & Knowledge","applicationDomains":["Cybersecurity","Robotics & Autonomous Systems"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity","Robotics"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.28053v1","pdf":"https://arxiv.org/pdf/2608.28053v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce CNeo-Bench, a benchmark of 4,759 such neologisms with reference definitions, organized into five top-level categories and nine subcategories by the linguistic mechanism behind each expression.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.28053"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"Chinese neologisms exploit diverse and unique linguistic mechanisms, such as phonetic substitution (e.g., 886 for ``bye-bye'') and visual character decomposition that are rare in other languages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T07:23:28.296315Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.28053v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"catalog_ab6fd1363b38386d","familyId":"catalog_family_ab6fd1363b38386d","name":"CNMO 2024","oneLine":"China Mathematical Olympiad 2024 - A challenging mathematics competition.","description":"China Mathematical Olympiad 2024 - A challenging mathematics competition.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cnmo-2024","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ab6fd1363b38386d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cnmo-2024"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cnmo-2024","url":"https://llm-stats.com/benchmarks/cnmo-2024","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_cocobench_0d6f9f05","familyId":"bmf_75548b7ea87a","name":"CoCoBench","oneLine":"Agent systems powered by multimodal large language models (MLLMs) have advanced rapidly in recent years, yet existing embodied-agent benchmarks still lack fine-grained diagnostics for multi-agent coordination.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Planning"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.28266v1","pdf":"https://arxiv.org/pdf/2608.28266v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"In this paper, we introduce CoCoBench, a construct-level benchmark for evaluating multi-agent embodied coordination in executable household tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.28266"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"Agent systems powered by multimodal large language models (MLLMs) have advanced rapidly in recent years, yet existing embodied-agent benchmarks still lack fine-grained diagnostics for multi-agent coordination.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T07:23:28.296315Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.28266v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_cocotree_73046ddd","familyId":"bmf_81fdc618d821","name":"COCOTree","oneLine":"COCOTree evaluates open tree-structured visual decomposition, segmenting images into hierarchical trees of visual components with unconstrained granularity. It includes over 21K images and 1.8M structural nodes, with 3.5K unique labels, and uses the Open Tree Quality (OTQ) metric for scoring.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22068","pdf":"https://arxiv.org/pdf/2605.22068","project":null,"code":"https://github.com/melonkick3090/COCOTree","data":null,"hfPaper":"https://huggingface.co/papers/2605.22068"},"evidence":{"snippet":"We release our dataset and benchmark code at https://github.com/melonkick3090/COCOTree.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.22068"},"ranking":{},"description":"COCOTree evaluates open tree-structured visual decomposition, segmenting images into hierarchical trees of visual components with unconstrained granularity. It includes over 21K images and 1.8M structural nodes, with 3.5K unique labels, and uses the Open Tree Quality (OTQ) metric for scoring.","whyItMatters":"Provides a standardized evaluation for a new task paradigm, enabling comparison across models on open vocabulary and long-tail visual structures, where existing benchmarks lack hierarchical decomposition metrics.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"033429e4f945fd68b342ba11b6424803e89a9109aec69387a01ee961b3d8faf0"},"motivation":"We formalize and enable the task of open tree decomposition, which segments an image into hierarchical trees of visual components with unconstrained granularity and flexibility.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.22068","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Anonymous","organizationType":"academic-lab","sourceUrl":"https://github.com/melonkick3090/COCOTree","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_cod10k-c_fc1b1fcc","familyId":"bmf_3d7b81bb10c7","name":"COD10K-C","oneLine":"COD10K-C evaluates camouflaged object detection models under 8 corruption types at 5 severity levels, yielding 40 conditions and 81,040 image pairs based on COD10K. Models are scored on Dice and other standard metrics for segmentation robustness.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Robustness"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02603","pdf":"https://arxiv.org/pdf/2606.02603","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02603"},"evidence":{"snippet":"We present COD10K-C, a corruption robustness benchmark based on COD10K.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02603"},"ranking":{},"description":"COD10K-C evaluates camouflaged object detection models under 8 corruption types at 5 severity levels, yielding 40 conditions and 81,040 image pairs based on COD10K. Models are scored on Dice and other standard metrics for segmentation robustness.","whyItMatters":"Standard camouflaged object detection benchmarks measure performance on clean images only, while real-world captures include blur, noise, weather, and compression artifacts. This benchmark quantifies robustness drops under such corruptions, enabling selection of models that degrade gracefully in practical conditions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ad4f73f8cfaff6e64c8aff53929fe65e5a172c6bd16ce20d29fd954fe6999de2"},"motivation":"Camouflaged object detection has improved substantially, but most standard benchmarks evaluate models only on clean images.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02603","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_coda-bench_a3ad2891","familyId":"bmf_7e9ed12d69ca","name":"CODA-BENCH","oneLine":"CODA-Bench is a benchmark for evaluating AI agents on data-intensive analytical tasks in a Linux sandbox with 1,009 tasks across 31 communities, requiring data discovery, code generation, and correct answers.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15300","pdf":"https://arxiv.org/pdf/2606.15300","project":"https://coda-bench.github.io/","code":"https://github.com/ruc-datalab/CoDA-Bench","data":"https://huggingface.co/datasets/RUC-DataLab/CoDA-Bench","hfPaper":"https://huggingface.co/papers/2606.15300"},"evidence":{"snippet":"In this paper, we bridge this gap by introducing CODA-BENCH, the first benchmark to jointly evaluate code and data intelligence in a data-intensive environment.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":13,"hfDailySubmittedAt":"2026-06-16T00:00:00.000Z","githubStars":44,"githubScope":"benchmark_repo","hfDatasetDownloads":431,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2606.15300"},"ranking":{"90d":{"score":51,"rank":61,"coverage":1.0,"confidence":"High","datasetDownloadRank":23,"datasetRankPopulation":66}},"description":"CODA-Bench is a benchmark for evaluating AI agents on data-intensive analytical tasks in a Linux sandbox with 1,009 tasks across 31 communities, requiring data discovery, code generation, and correct answers.","whyItMatters":"Existing benchmarks evaluate code or data capabilities in isolation. CODA-Bench jointly evaluates both, reflecting real development scenarios with complex file systems and large-scale data, revealing gaps in agentic data intelligence.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bc53bacf180f2fd3e57bb930c67527a8ab71526a260aaa2104150d7856741a38"},"motivation":"Advanced agents are increasingly demonstrating the potential to operate as autonomous engineers, creating a growing demand for evaluation benchmarks that capture the complexity of real-world development.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026","evidence":"Accepted at ICML 2026. 37 pages, 11 figures. Project page: https://coda-bench.github.io/ Code: https://github.com/ruc-datalab/CoDA-Bench Data: https://huggingface.co/datasets/RUC-DataLab/CoDA-Bench","evidenceUrl":"https://arxiv.org/abs/2606.15300","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026","reviewStatus":"accepted","decisionRaw":"Accepted at ICML 2026. 37 pages, 11 figures. Project page: https://coda-bench.github.io/ Code: https://github.com/ruc-datalab/CoDA-Bench Data: https://huggingface.co/datasets/RUC-DataLab/CoDA-Bench","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.15300","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at ICML 2026. 37 pages, 11 figures. Project page: https://coda-bench.github.io/ Code: https://github.com/ruc-datalab/CoDA-Bench Data: https://huggingface.co/datasets/RUC-DataLab/CoDA-Bench","level":"author-claim"}]}],"publishers":[{"name":"RUC-DataLab","organizationType":"academic-lab","sourceUrl":"https://github.com/ruc-datalab/CoDA-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_8fe6d62e6bb429f5","familyId":"catalog_family_8fe6d62e6bb429f5","name":"Code Migration","oneLine":"Can language models reimplement real-world programs in another language?","description":"Can language models reimplement real-world programs in another language?","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/code-migration","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8fe6d62e6bb429f5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/codemigration"}],"catalogSources":[{"catalog":"benchlm","sourceId":"codeMigration","url":"https://benchlm.ai/benchmarks/codemigration","paperUrl":"https://www.vals.ai/benchmarks/code-migration","year":"2026","fullName":"Vals Code Migration","format":"Accuracy score","tasks":"Real-world program reimplementation in another language","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_codeassay_bb1ca7aa","familyId":"bmf_30e11e735fce","name":"CodeAssay","oneLine":"CodeAssay is a benchmark of 185 Python tasks across ten software-engineering categories with audited ground truth, public and hidden tests, and code-property measures. It evaluates LLM code generation correctness and other code properties.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation"],"topics":["Code"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03535","pdf":"https://arxiv.org/pdf/2608.03535","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03535"},"evidence":{"snippet":"We present CodeAssay, a taxonomy-first benchmark of 185 Python tasks across ten software-engineering categories.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03535"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CodeAssay is a benchmark of 185 Python tasks across ten software-engineering categories with audited ground truth, public and hidden tests, and code-property measures. It evaluates LLM code generation correctness and other code properties.","whyItMatters":"Provides a reproducible basis for evaluating LLM-generated code with validated ground truth and multiple metrics, enabling evidence-based model selection in software development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f1d010099065690cca44a1f81bf2420f8a06c01f23ebde611e338cbdde5c3228"},"motivation":"Large Language Models are increasingly evaluated for code generation using test-based benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03535","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_e13a9b6e8ac54db6","familyId":"catalog_family_e13a9b6e8ac54db6","name":"Codeforces","oneLine":"A competitive programming benchmark using problems from the CodeForces platform. The benchmark evaluates code generation capabilities of LLMs on algorithmic problems with difficulty ratings ranging from 800 to 2400. Problems cover diverse algorithmic categories including dynamic programming, graph algorithms, data structures, and mathematical problems with standardized evaluation through direct platform submission.","description":"A competitive programming benchmark using problems from the CodeForces platform. The benchmark evaluates code generation capabilities of LLMs on algorithmic problems with difficulty ratings ranging from 800 to 2400. Problems cover diverse algorithmic categories including dynamic programming, graph algorithms, data structures, and mathematical problems with standardized evaluation through direct platform submission.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e13a9b6e8ac54db6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/codeforces"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/codeforces"}],"catalogSources":[{"catalog":"benchlm","sourceId":"codeforces","url":"https://benchlm.ai/benchmarks/codeforces","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"Codeforces Rating","format":"Rating","tasks":"Competitive programming contests","successorKey":null},{"catalog":"llm-stats","sourceId":"codeforces","url":"https://llm-stats.com/benchmarks/codeforces","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","math","reasoning"],"catalogModelCount":17,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_codegenbench_b6866f0c","familyId":"bmf_25ab5218d3f7","name":"CodegenBench","oneLine":"CodegenBench evaluates LLM-generated parallel code across x86_64, Sunway, and Kunpeng architectures using BLAS routines and specialized kernels. Scoring measures efficiency on each platform.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04023","pdf":"https://arxiv.org/pdf/2606.04023","project":"https://anonymous.4open.science/r/CodegenBench-EDE1/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04023"},"evidence":{"snippet":"To bridge this gap, we introduce CodegenBench, a comprehensive benchmark suite designed to evaluate the generation of efficient parallel code across three distinct hardware platforms: x86_64, Sunway, and Kunpeng.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04023"},"ranking":{},"description":"CodegenBench evaluates LLM-generated parallel code across x86_64, Sunway, and Kunpeng architectures using BLAS routines and specialized kernels. Scoring measures efficiency on each platform.","whyItMatters":"Addresses the gap in evaluating code generation for CPU-oriented HPC platforms, revealing cross-platform generalization limitations that matter for deploying LLMs in supercomputing contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"13e8dba48bbcffc07c2b92d947ad8cb04d4c16124dcef3c28a377910ee231e54"},"motivation":"While large language models (LLMs) have been extensively evaluated on code generation tasks for general-purpose programming and GPU-accelerated environments (e.g., PyTorch, CUDA), their capabilities in CPU-oriented high-performance computing (HPC) across diverse architectures remain underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04023","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_codegolf-bench_db502fb1","familyId":"bmf_37309a2fff0f","name":"CodeGolf Bench","oneLine":"CodeGolf Bench evaluates concise code generation across 60 programming languages, using code golf platform problems and human performance baselines for scoring.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation"],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30394","pdf":"https://arxiv.org/pdf/2605.30394","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30394"},"evidence":{"snippet":"This paper introduces Code Bench, a benchmark capable of evaluating Large Language Models (LLMs) concise code generation abilities in 60 programming languages.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30394"},"ranking":{},"description":"CodeGolf Bench evaluates concise code generation across 60 programming languages, using code golf platform problems and human performance baselines for scoring.","whyItMatters":"Existing code benchmarks focus on correctness, not efficiency or conciseness. CodeGolf Bench offers a unique measure of LLM ability to produce minimal solutions, complementing standard code generation evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8f793e96d2bee619d9b31bcc326a46c3b8551a79e7d96ee2f6773f1777fa0a7e"},"motivation":"This paper introduces Code Bench, a benchmark capable of evaluating Large Language Models (LLMs) concise code generation abilities in 60 programming languages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30394","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_e261ed54dd2f1c07","familyId":"catalog_family_e261ed54dd2f1c07","name":"Codegolf v2.2","oneLine":"Codegolf v2.2 benchmark","description":"Codegolf v2.2 benchmark","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/codegolf-v2.2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e261ed54dd2f1c07"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/codegolf-v2.2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"codegolf-v2.2","url":"https://llm-stats.com/benchmarks/codegolf-v2.2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_coffeebench_22526372","familyId":"bmf_43631014a768","name":"CoffeeBench","oneLine":"CoffeeBench evaluates LLM agents as a coffee roaster in a 90-day multi-agent economy with fixed reference agents, measuring cumulative net income through autonomous communication, negotiation, and transactions.","area":"Language & Knowledge","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Logistics"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.16613","pdf":"https://arxiv.org/pdf/2606.16613","project":null,"code":"https://github.com/SakanaAI/CoffeeBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.16613"},"evidence":{"snippet":"We introduce CoffeeBench, a benchmark for evaluating LLM agents in a long-horizon multi-agent economy composed of heterogeneous firms.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":10,"hfDailySubmittedAt":"2026-06-26T00:00:00.000Z","githubStars":32,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16613"},"ranking":{"90d":{"score":48,"rank":87,"coverage":0.7,"confidence":"Medium"}},"description":"CoffeeBench evaluates LLM agents as a coffee roaster in a 90-day multi-agent economy with fixed reference agents, measuring cumulative net income through autonomous communication, negotiation, and transactions.","whyItMatters":"Existing benchmarks often focus on single agents in static environments, whereas CoffeeBench captures long-horizon multi-agent economic interactions. It provides a scoring contract based on net income, enabling comparison of agent economic decision-making and revealing failure modes like idle-drift.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b4cb6b5ad9df185d28eb61b7775d1b7d7587997c1f20c31ff01e45c2db9c5c89"},"motivation":"As LLM agents become capable of increasingly long-horizon tasks, evaluating their performance in economic systems is becoming increasingly important.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.16613","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Sakana AI","organizationType":"company-research-lab","sourceUrl":"https://github.com/SakanaAI/CoffeeBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_cogcanvas_d3de924a","familyId":"bmf_5fd28976da11","name":"CogCanvas","oneLine":"CogCanvas is a benchmark for multi-subject reference-based image generation, with 1,952 curated reference images, 1,361 compositional prompts, and a unified six-axis evaluation protocol. It includes tasks for reference-based generation, text-to-image composition, and reference retrieval, with metrics BG-Sim and Attr-VQA.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15867","pdf":"https://arxiv.org/pdf/2606.15867","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.15867"},"evidence":{"snippet":"We introduce CogCanvas, a benchmark of 1,952 curated reference images spanning 100 celebrity identities, 115 distinctive objects and fashion items, and 29 real-world background scenes including landmarks, from which we construct 1,361 compositional prompts covering 2-5 person group sizes.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15867"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CogCanvas is a benchmark for multi-subject reference-based image generation, with 1,952 curated reference images, 1,361 compositional prompts, and a unified six-axis evaluation protocol. It includes tasks for reference-based generation, text-to-image composition, and reference retrieval, with metrics BG-Sim and Attr-VQA.","whyItMatters":"CogCanvas addresses the gap in jointly evaluating multi-identity, object binding, background grounding, and spatial plausibility in image generation. It provides a comprehensive benchmark for assessing composition capabilities across group sizes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"399d38ad19fb4377b9470179b792d6a403db3f0c94795b3db9317e1669b49650"},"motivation":"Multi-subject reference-based image generation requires jointly preserving multiple human identities, binding per-person objects and fashion items, and respecting a specified background scene, a regime where current diffusion models remain brittle.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15867","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_cogmanip_d6109b16","familyId":"bmf_b16b8f702555","name":"CogManip","oneLine":"Evaluates 15 manipulative behavior strategies in 1,000 multi-turn LLM interaction scenarios.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06099","pdf":"https://arxiv.org/pdf/2606.06099","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.06099"},"evidence":{"snippet":"We introduce CogManip, a comprehensive benchmark that evaluates 15 manipulation strategy risks across 1,000 multi-turn interaction scenarios, validated by human experts.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06099"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates 15 manipulative behavior strategies in 1,000 multi-turn LLM interaction scenarios.","whyItMatters":"Could support safety auditing of dynamic covert manipulation, but lacks public access to scenarios and scoring details.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"18e890a3d4f2d47c6d46a5baa226ff2483feddfcbe3129e396cde4fe79f8c742"},"motivation":"Whether Large Language Models (LLMs) exhibit covert psychological manipulation in complex human-AI interactions has garnered increasing safety concerns.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.06099","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_a0b527c3bcba968d","familyId":"catalog_family_a0b527c3bcba968d","name":"Cohere Agentic Question Answering","oneLine":"Cohere's internal North evaluation for measuring how well a model answers enterprise questions using MCP-connected cloud file systems. Scores are reported with LLM-as-a-judge techniques.","description":"Cohere's internal North evaluation for measuring how well a model answers enterprise questions using MCP-connected cloud file systems. Scores are reported with LLM-as-a-judge techniques.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Question Answering","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cohere-agentic-question-answering","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a0b527c3bcba968d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cohere-agentic-question-answering"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cohere-agentic-question-answering","url":"https://llm-stats.com/benchmarks/cohere-agentic-question-answering","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["question answering","reasoning","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_e9ba1beedb22161e","familyId":"catalog_family_e9ba1beedb22161e","name":"Cohere Data Analysis","oneLine":"Cohere's internal North evaluation for measuring a model's ability to perform data science tasks over uploaded spreadsheets. Scores are reported with LLM-as-a-judge techniques.","description":"Cohere's internal North evaluation for measuring a model's ability to perform data science tasks over uploaded spreadsheets. Scores are reported with LLM-as-a-judge techniques.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Data Analysis","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cohere-data-analysis","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e9ba1beedb22161e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cohere-data-analysis"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cohere-data-analysis","url":"https://llm-stats.com/benchmarks/cohere-data-analysis","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["data analysis","reasoning","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_b0e2f30b12a4f61b","familyId":"catalog_family_b0e2f30b12a4f61b","name":"Cohere Memory Usage Quality","oneLine":"Cohere's internal North evaluation for measuring how well an agent uses information from North's memory system across sessions. Scores are reported with LLM-as-a-judge techniques.","description":"Cohere's internal North evaluation for measuring how well an agent uses information from North's memory system across sessions. Scores are reported with LLM-as-a-judge techniques.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Memory","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cohere-memory-usage-quality","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b0e2f30b12a4f61b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cohere-memory-usage-quality"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cohere-memory-usage-quality","url":"https://llm-stats.com/benchmarks/cohere-memory-usage-quality","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["memory","reasoning","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_coinve-bench_5d91b406","familyId":"bmf_05d84e5ad74c","name":"CoinVE-Bench","oneLine":"CoinVE-Bench provides 361 multi-instruction test cases for compositional video editing, with four-dimensional evaluation metrics covering instruction following, editing accuracy, visual quality, and temporal consistency.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":0.75,"links":{"report":"https://arxiv.org/abs/2608.17566","pdf":"https://arxiv.org/pdf/2608.17566","project":"https://coinve200k.github.io","code":"https://github.com/coinve200k/CoinVE-200K","data":"https://huggingface.co/datasets/FireCRT/CoinVE-200K","hfPaper":null},"evidence":{"snippet":"We also introduce CoinVE-Bench, a benchmark for compositional-instruction video editing across diverse subjects, operation types, and instruction complexities.","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":15,"hfDailySubmittedAt":"2026-08-19T00:00:00.000Z","githubStars":29,"githubScope":"benchmark_repo","hfDatasetDownloads":11732,"hfDatasetLikes":11},"source":{"type":"arxiv","id":"2608.17566"},"ranking":{"30d":{"score":58,"rank":13,"coverage":1.0,"confidence":"High","datasetDownloadRank":1,"datasetRankPopulation":30},"90d":{"score":56,"rank":31,"coverage":1.0,"confidence":"High","datasetDownloadRank":3,"datasetRankPopulation":66}},"description":"CoinVE-Bench provides 361 multi-instruction test cases for compositional video editing, with four-dimensional evaluation metrics covering instruction following, editing accuracy, visual quality, and temporal consistency.","whyItMatters":"CoinVE-Bench offers a dedicated evaluation set for compositional instruction-guided video editing, addressing gaps in existing single-operation benchmarks and enabling standardized model comparison.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"3c94c0c4a2ec5a98a11adb7bebf26ebf923555617179e1c77ba68dcbf536cbd1"},"motivation":"The quality and diversity of instruction-based video editing datasets are steadily improving, yet existing datasets mainly focus on single editing operations and fall short in supporting compositional instruction-guided video editing.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"CoinVE-Bench is explicitly introduced as a benchmark in the abstract, has a public GitHub repository and Hugging Face dataset, and defines evaluation metrics for the task.","canonicalNameSource":"abstract","canonicalNameEvidence":"We also introduce CoinVE-Bench, a benchmark for compositional-instruction video editing across diverse subjects, operation types, and instruction complexities."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17566","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":60,"confidence":"Medium","horizon":"7d","reason":"The benchmark accompanies a large-scale dataset and model release, with public repositories, likely to attract attention from video editing and multimodal model communities."},"evaluationMode":"public_reusable","publishers":[{"name":"Tencent Smart Creation Platform Department","organizationType":"company-research-lab","sourceUrl":"https://github.com/coinve200k/CoinVE-200K","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_collabbench_1ee76547","familyId":"bmf_85f88426d607","name":"CollabBench","oneLine":"Evaluates collaborative ability of LLM agents in cooperative game environments with diverse player profiles and proactive engagement.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05793","pdf":"https://arxiv.org/pdf/2606.05793","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05793"},"evidence":{"snippet":"To this end, this paper proposes CollabBench, a benchmark for evaluating and training collaborative agents in cooperative games.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05793"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates collaborative ability of LLM agents in cooperative game environments with diverse player profiles and proactive engagement.","whyItMatters":"Could fill gap in grounded collaborative benchmarks, but lacks public evidence of evaluation protocol or artifacts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"59de895f03c1c23e27dbc550f2dfb294b8b32d5fd9f3abca9175eb43f1e09df2"},"motivation":"While LLM-based agents excel at individual tasks, effective collaboration with realistic human partners remains challenging.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026","evidence":"Accepted by ICML 2026","evidenceUrl":"https://arxiv.org/abs/2606.05793","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026","reviewStatus":"accepted","decisionRaw":"Accepted by ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.05793","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by ICML 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_fcb758b78074e827","familyId":"catalog_family_fcb758b78074e827","name":"COLLIE","oneLine":"COLLIE is a grammar-based framework for systematic construction of constrained text generation tasks. It allows specification of rich, compositional constraints across diverse generation levels and modeling challenges including language understanding, logical reasoning, and semantic planning. The COLLIE-v1 dataset contains 2,080 instances across 13 constraint structures.","description":"COLLIE is a grammar-based framework for systematic construction of constrained text generation tasks. It allows specification of rich, compositional constraints across diverse generation levels and modeling challenges including language understanding, logical reasoning, and semantic planning. The COLLIE-v1 dataset contains 2,080 instances across 13 constraint structures.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning","Writing"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/collie","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fcb758b78074e827"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/collie"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"collie","url":"https://llm-stats.com/benchmarks/collie","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning","writing"],"catalogModelCount":10,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_colosseum-v2_833bffbd","familyId":"bmf_0060dc974fcc","name":"Colosseum V2","oneLine":"Simulation benchmark for VLA generalization with 28 tasks across 13 categories and two robot morphologies, using ManiSkill.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27759","pdf":"https://arxiv.org/pdf/2605.27759","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27759"},"evidence":{"snippet":"To systematically study this gap, we introduce Colosseum V2, a large-scale simulation benchmark for evaluating VLA generalization in robot learning across diverse conditions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27759"},"ranking":{},"description":"Simulation benchmark for VLA generalization with 28 tasks across 13 categories and two robot morphologies, using ManiSkill.","whyItMatters":"VLA models often fail under distribution shifts. Colosseum V2 provides standardized in/out-domain evaluation for reproducible comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f91ed8f06b456df929296bd62c360d5f4e98d71329f8ce91c1d3019f874bd626"},"motivation":"Vision-Language-Action (VLA) models demonstrate promising generalization in robotic manipulation, driven by advances in large-scale vision and language pre-training.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27759","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Colosseum V2 Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2605.27759","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_combench_b27be297","familyId":"bmf_29e41fd248d6","name":"ComBench","oneLine":"ComBench evaluates large language models on 100 human-annotated Olympiad-level combinatorics problems, split into 50 analysis-centric and 50 construction-centric tasks. Scoring combines rubric-guided proof grading with deterministic verification of construction outputs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.10479","pdf":"https://arxiv.org/pdf/2606.10479","project":"https://simplified-reasoning.github.io/ComBench/docs/","code":"https://github.com/Simplified-Reasoning/ComBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.10479"},"evidence":{"snippet":"We introduce ComBench, an Olympiad-level combinatorics benchmark for evaluating and diagnosing the combinatorial reasoning capabilities of large language models.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":20,"hfDailySubmittedAt":"2026-06-11T00:00:00.000Z","githubStars":14,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.10479"},"ranking":{"90d":{"score":43,"rank":126,"coverage":0.7,"confidence":"Medium"}},"description":"ComBench evaluates large language models on 100 human-annotated Olympiad-level combinatorics problems, split into 50 analysis-centric and 50 construction-centric tasks. Scoring combines rubric-guided proof grading with deterministic verification of construction outputs.","whyItMatters":"ComBench fills a gap in evaluating creative and rigorous combinatorial reasoning at the Olympiad level. It provides separate scores for proof quality and construction validity, helping diagnose where models diverge in these capabilities, and enables fine-grained comparison of frontier models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b0e3f91093bc15da60e71ed07dcd7dcac15cba99db7a2fa710aecffbd7e374ff"},"motivation":"Combinatorics is central to Olympiad-level mathematical problem solving, requiring deep discrete reasoning, creative constructions, and rigorous structural insight.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.10479","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Simplified Reasoning","organizationType":"academic-lab","sourceUrl":"https://github.com/Simplified-Reasoning/ComBench","role":"benchmark-publisher"}],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_combeval_06893a13","familyId":"bmf_1b00e4f89969","name":"CombEval","oneLine":"CombEval evaluates combinatorial counting abilities of large language models using problems generated from typed Cofola specifications, with solver-verified answers. It supports systematic variation of object types, entity scales, constraint counts, and reasoning depth.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19788","pdf":"https://arxiv.org/pdf/2606.19788","project":null,"code":"https://github.com/YuxuZhou-CN/combination-problem-generation","data":null,"hfPaper":"https://huggingface.co/papers/2606.19788"},"evidence":{"snippet":"We present CombEval, a dynamic benchmark for evaluating combinatorial counting in large language models.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19788"},"ranking":{"90d":{"score":23,"rank":378,"coverage":0.55,"confidence":"Low"}},"description":"CombEval evaluates combinatorial counting abilities of large language models using problems generated from typed Cofola specifications, with solver-verified answers. It supports systematic variation of object types, entity scales, constraint counts, and reasoning depth.","whyItMatters":"Addresses the gap in dynamic evaluation of combinatorial reasoning, providing controlled generation and exact verification to diagnose model failures in counting tasks, useful for targeted improvements.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dd040b068a53163a0e71df6f165ac27d2568013bf9d9a183de7cbdb0c838f090"},"motivation":"We present CombEval, a dynamic benchmark for evaluating combinatorial counting in large language models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19788","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"YuxuZhou-CN","organizationType":"community","sourceUrl":"https://github.com/YuxuZhou-CN/combination-problem-generation","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_comboshoppingbench_78836e04","familyId":"bmf_9c63aab8e01d","name":"ComboShoppingBench","oneLine":"ComboShoppingBench is a benchmark for agentic basket shopping with coupons, evaluating LLM agents on constructing feasible baskets with budget and coupon constraints in a simulated environment.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09282","pdf":"https://arxiv.org/pdf/2608.09282","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09282"},"evidence":{"snippet":"We introduce ComboShoppingBench, an agentic shopping benchmark for open-ended yet verifiable basket construction in a simulated commerce and takeout environment.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09282"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ComboShoppingBench is a benchmark for agentic basket shopping with coupons, evaluating LLM agents on constructing feasible baskets with budget and coupon constraints in a simulated environment.","whyItMatters":"It addresses the gap in evaluating agents for combinatorial shopping tasks, which require joint reasoning about compatibility, availability, and constraints.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"18268737539c82cec70a4fc624a8fc044c7cf86e22c84acf8730489620fd3b7c"},"motivation":"Real-world shopping often requires constructing a basket of complementary items rather than retrieving a single product.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09282","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_comedbench_50407db9","familyId":"bmf_572ffe3d9a69","name":"CoMedBench","oneLine":"CoMedBench evaluates synthetic medical data fidelity and downstream utility across 37 dataset-task pairs from seven public sources, using a common clinical-validity framework and shared training and evaluation engine.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.LG"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12805","pdf":"https://arxiv.org/pdf/2608.12805","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12805"},"evidence":{"snippet":"We introduce CoMedBench, a reproducible benchmark that evaluates a family of generators under a common clinical-validity framework and one shared training and evaluation engine, spanning static tabular and temporal downstream tasks on established critical-care datasets.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12805"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CoMedBench evaluates synthetic medical data fidelity and downstream utility across 37 dataset-task pairs from seven public sources, using a common clinical-validity framework and shared training and evaluation engine.","whyItMatters":"Provides a reproducible benchmark for comparing synthetic data generators across multiple datasets and tasks, addressing the lack of comprehensive evaluation in prior studies and aiding in deciding when synthetic data is viable for model development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3dee9274de668d7383680dcaf1e8bc0c99f5dcd98d549e0d8e80998c575d1493"},"motivation":"Access to clinical data is essential for developing reliable healthcare machine learning systems, but direct use of electronic health records is constrained by privacy regulation, institutional review, data-use agreements, and the risk of re-identification.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12805","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_comet-bench_355dabd3","familyId":"bmf_2bde53728ad5","name":"CoMET-Bench","oneLine":"CoMET-Bench is a benchmark for conditional multi-event temporal grounding in long-form video, with 2,789 queries over 600 videos and a unified evaluation protocol including counting, grounding, and negative-query recognition.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15320","pdf":"https://arxiv.org/pdf/2606.15320","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.15320"},"evidence":{"snippet":"We introduce CoMET-Bench for Conditional Multi-Event Temporal Grounding in long-form video, comprising 2789 queries over 600 videos averaging 33.8 minutes across five real-world domains, with each query composed from 4 temporal conditions, 3 spatial conditions, and a dedicated negative-query subset.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15320"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CoMET-Bench is a benchmark for conditional multi-event temporal grounding in long-form video, with 2,789 queries over 600 videos and a unified evaluation protocol including counting, grounding, and negative-query recognition.","whyItMatters":"Real-world video grounding requires localizing every event satisfying compositional conditions, which existing benchmarks do not jointly handle. This benchmark introduces Rejection-F1 to prevent trivial gaming and exposes gaps in current methods.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"58c5389c132002ab78491d9e7721fe06ff02127749af1a073657480c0b943e1d"},"motivation":"Multimodal large language models have made rapid progress in video temporal grounding, yet real-world applications routinely require localizing every event that satisfies compositional temporal and spatial conditions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15320","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_comex_9077da26","familyId":"bmf_34d8cbd69dec","name":"COMEX","oneLine":"COMEX is a benchmark for explainable aesthetic image cropping containing 33,161 quadruples of expanded image, crop box, composition category, and composition-grounded explanation. It supports joint evaluation of crop localization, composition understanding, and explanation generation.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07570","pdf":"https://arxiv.org/pdf/2608.07570","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07570"},"evidence":{"snippet":"To support this setting, we introduce COMEX, a new benchmark built through image expansion and an IO-reversal pipeline.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07570"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"COMEX is a benchmark for explainable aesthetic image cropping containing 33,161 quadruples of expanded image, crop box, composition category, and composition-grounded explanation. It supports joint evaluation of crop localization, composition understanding, and explanation generation.","whyItMatters":"Provides a structured evaluation of crop-and-explain methods, enabling comparison across models on composition-grounded reasoning and explanation generation, addressing the gap of evaluating explainability in image cropping.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"778832c8b2ffa61e32f6c420cf2ecdbb70d67cbeb9fdd63b3a78b3fc98a3f1ed"},"motivation":"Explainable aesthetic image cropping requires not only localizing a visually pleasing crop but also explaining why it is preferred.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07570","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_838d14d939d6a079","familyId":"catalog_family_838d14d939d6a079","name":"Common Voice 15","oneLine":"Common Voice is a massively-multilingual collection of transcribed speech intended for speech technology research and development. Version 15.0 contains 28,750 recorded hours across 114 languages, consisting of crowdsourced voice recordings with corresponding transcriptions.","description":"Common Voice is a massively-multilingual collection of transcribed speech intended for speech technology research and development. Version 15.0 contains 28,750 recorded hours across 114 languages, consisting of crowdsourced voice recordings with corresponding transcriptions.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Speech To Text","Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/common-voice-15","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_838d14d939d6a079"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/common-voice-15"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"common-voice-15","url":"https://llm-stats.com/benchmarks/common-voice-15","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","speech to text","audio"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_1f306761f559c50a","familyId":"catalog_family_1f306761f559c50a","name":"CommonSenseQA","oneLine":"CommonSenseQA is a multiple-choice question answering dataset that requires different types of commonsense knowledge to predict correct answers. It contains 12,102 questions with one correct answer and four distractors, designed to test semantic reasoning and conceptual relationships. Questions are created based on ConceptNet concepts and require prior world knowledge for accurate reasoning.","description":"CommonSenseQA is a multiple-choice question answering dataset that requires different types of commonsense knowledge to predict correct answers. It contains 12,102 questions with one correct answer and four distractors, designed to test semantic reasoning and conceptual relationships. Questions are created based on ConceptNet concepts and require prior world knowledge for accurate reasoning.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/commonsenseqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1f306761f559c50a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/commonsenseqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"commonsenseqa","url":"https://llm-stats.com/benchmarks/commonsenseqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_communityfact_765beb88","familyId":"bmf_e432ba8d77f7","name":"CommunityFact","oneLine":"CommunityFact evaluates misinformation detection on 15,992 claims across five languages and two domains, using accuracy against human Community Notes ratings as the scoring metric.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30241","pdf":"https://arxiv.org/pdf/2605.30241","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30241"},"evidence":{"snippet":"We introduce CommunityFact, a refreshable benchmark for misinformation detection in the wild, with three major goals: coverage, granularity, and redistributability.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30241"},"ranking":{},"description":"CommunityFact evaluates misinformation detection on 15,992 claims across five languages and two domains, using accuracy against human Community Notes ratings as the scoring metric.","whyItMatters":"Static benchmarks fail to capture dynamic, multilingual misinformation settings. CommunityFact provides a refreshable benchmark to assess model reliability in real-world verification contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ac02f4873235b73e415ed66283bea76506aee0d969e86335a0f5fc642a1e44fc"},"motivation":"Misinformation verification increasingly occurs in public, fast-moving, and multilingual online settings, where static benchmarks provide an incomplete measure of model reliability.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30241","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_companionbench_b876ca73","familyId":"bmf_fa6e96451701","name":"CompanionBench","oneLine":"CompanionBench is an interactive bilingual benchmark for AI emotional companionship, grounding scenarios and a user simulator in de-identified real-world data. It evaluates ten capabilities derived from 25 theories, using a rubric and deterministic disclosure measure.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.02046","pdf":"https://arxiv.org/pdf/2608.02046","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.02046"},"evidence":{"snippet":"We introduce CompanionBench, an interactive bilingual benchmark.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02046"},"ranking":{"30d":{"score":39,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"CompanionBench is an interactive bilingual benchmark for AI emotional companionship, grounding scenarios and a user simulator in de-identified real-world data. It evaluates ten capabilities derived from 25 theories, using a rubric and deterministic disclosure measure.","whyItMatters":"LLM companions are deployed at scale but poorly evaluated. CompanionBench provides a reproducible, theory-anchored benchmark with real-world grounding, offering granular capability assessment and addressing judge biases.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"34da21590388dc063177ffd629f6d7635a793effad0796df39d51dba051bc8b8"},"motivation":"LLM companions are deployed at scale in personally consequential settings, yet poorly evaluated.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02046","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_company-mental-model-benchmark_e52dec26","familyId":"bmf_2a0a50cba0bc","name":"Company Mental Model Benchmark","oneLine":"Compares three evidence-to-thesis treatments for LLM-based company analysis across eight companies, scored by blind reviewers against reference models on identifying dominant compounding mechanisms.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Software & Cloud","Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/0xboyu/company-mental-model-benchmark","pdf":null,"project":null,"code":"https://github.com/0xboyu/company-mental-model-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"company-mental-model-benchmark Research benchmark for testing explicit company economic-model representations against direct LLM synthesis.","reasonCodes":["discovered via github","benchmark term in abstract","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:0xboyu/company-mental-model-benchmark"},"ranking":{"30d":{"score":23,"rank":127,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":331,"coverage":0.55,"confidence":"Low"}},"description":"Compares three evidence-to-thesis treatments for LLM-based company analysis across eight companies, scored by blind reviewers against reference models on identifying dominant compounding mechanisms.","whyItMatters":"Provides a falsifiable harness to test whether explicit economic model representations improve LLM reasoning for investment analysis, with a clear protocol and rubric.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"62e32e145270cfc13200a3328a34a3af4eb9b85eee7d6eaa1af457e66f62a594"},"motivation":"company-mental-model-benchmark Research benchmark for testing explicit company economic-model representations against direct LLM synthesis.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/0xboyu/company-mental-model-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":25,"confidence":"Low","horizon":"7d","reason":"The niche focus on company mental models and lack of executed results limits expected early attention, though the benchmark design may interest LLM evaluation researchers."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"specific"},{"id":"catalog_3e912c727f61a95c","familyId":"catalog_family_3e912c727f61a95c","name":"ComplexFuncBench","oneLine":"ComplexFuncBench is a benchmark designed to evaluate large language models' capabilities in handling complex function calling scenarios. It encompasses multi-step and constrained function calling tasks that require long-parameter filling, parameter value reasoning, and managing contexts up to 128k tokens. The benchmark includes 1,000 samples across five real-world scenarios.","description":"ComplexFuncBench is a benchmark designed to evaluate large language models' capabilities in handling complex function calling scenarios. It encompasses multi-step and constrained function calling tasks that require long-parameter filling, parameter value reasoning, and managing contexts up to 128k tokens. The benchmark includes 1,000 samples across five real-world scenarios.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Long Context","Reasoning","Structured Output","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/complexfuncbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3e912c727f61a95c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/complexfuncbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"complexfuncbench","url":"https://llm-stats.com/benchmarks/complexfuncbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","structured output","tool calling"],"catalogModelCount":7,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling","Long Context & Memory"],"domainScope":"general"},{"id":"bm_complexityworld_8718400b","familyId":"bmf_a96f895480e9","name":"ComplexityWorld","oneLine":"ComplexityWorld is a benchmark of 390 visual decision-making tasks across 39 worlds, scored by an executable verifier. It evaluates VLMs on tasks requiring global constraint satisfaction.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07584","pdf":"https://arxiv.org/pdf/2608.07584","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07584"},"evidence":{"snippet":"We introduce COMPLEXITYWORLD, a benchmark of 390 tasks across 39 domain-inspired visual worlds and 29 decision categories.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07584"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ComplexityWorld is a benchmark of 390 visual decision-making tasks across 39 worlds, scored by an executable verifier. It evaluates VLMs on tasks requiring global constraint satisfaction.","whyItMatters":"It targets a persistent visual-to-decision bottleneck in VLMs, but the lack of public artifacts and scoring details hinders independent verification.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"14061cd27b8e0b24c6793af467a056e91f670972e423bb458b5a211a38acd4aa"},"motivation":"Vision-language models (VLMs) have made rapid progress in visual perception and increasingly support real-world tasks that depend on images.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07584","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_compskillbench_264db8aa","familyId":"bmf_6de15eac705a","name":"CompSkillBench","oneLine":"CompSkillBench evaluates compositional skill routing for LLM agents: given a user query and a library of 2,209 real MCP server skills across 24 categories, systems must decompose the query into sub-tasks, retrieve a skill per sub-task, and produce an executable plan. Scoring uses step-level category recall and decomposition accuracy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18051","pdf":"https://arxiv.org/pdf/2606.18051","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18051"},"evidence":{"snippet":"To support evaluation, we introduce CompSkillBench, a benchmark of 300 compositional queries over 2,209 real MCP server skills spanning 24 functional categories, sourced from the public MCP ecosystem.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18051"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"CompSkillBench evaluates compositional skill routing for LLM agents: given a user query and a library of 2,209 real MCP server skills across 24 categories, systems must decompose the query into sub-tasks, retrieve a skill per sub-task, and produce an executable plan. Scoring uses step-level category recall and decomposition accuracy.","whyItMatters":"Real-world agent tasks often require composing multiple tools, but existing benchmarks emphasize single-skill selection. CompSkillBench provides a reusable, ecosystem-grounded benchmark to measure decomposition quality and retrieval in a compositional setting, helping developers identify bottlenecks in agent pipelines—particularly the critical role of task decomposition before retrieval.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"553c75df1ad6e3823e7c6253733a0e05ffa540f0c43794b319282f8b7d7dbac3"},"motivation":"LLM agents increasingly rely on external skills -- reusable tool specifications -- but real-world tasks often require composing multiple skills, not just selecting one.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18051","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SkillWeaver Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.18051","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_conceptedit-bench_65e0c6a3","familyId":"bmf_1b817470614a","name":"ConceptEdit-Bench","oneLine":"A granular evaluation suite for image editing across over 1,000 fine-grained edit concepts, designed to diagnose model capabilities across real-world scenarios.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.16812","pdf":"https://arxiv.org/pdf/2608.16812","project":null,"code":"https://github.com/inclusionAI/ConceptEdit","data":null,"hfPaper":"https://huggingface.co/papers/2608.16812"},"evidence":{"snippet":"Finally, we present ConceptEdit-Bench, a granular evaluation suite designed to diagnose model capabilities across a vast array of real-world scenarios.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":50,"hfDailySubmittedAt":"2026-08-25T00:00:00.000Z","githubStars":36,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.16812"},"ranking":{"30d":{"score":58,"rank":12,"coverage":0.85,"confidence":"High"},"90d":{"score":53,"rank":49,"coverage":0.7,"confidence":"Medium"}},"description":"A granular evaluation suite for image editing across over 1,000 fine-grained edit concepts, designed to diagnose model capabilities across real-world scenarios.","whyItMatters":"Image editing lacks benchmarks that test fine-grained concept editing; this suite provides a way to assess model capabilities across a wide range of scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-25T14:49:41.161722Z","inputHash":"33d53c3a59ed73ba21aae9a89218327da460609830f8d1f500419b3f13c11ee6"},"motivation":"Existing image editing frameworks predominantly follow the training paradigm of text-to-image diffusion models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.16812","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_87ed6a32388e715f","familyId":"catalog_family_87ed6a32388e715f","name":"Conceptual Reasoning","oneLine":"Tests whether model judgments rank argumentative critiques in the same order as expert human ratings across philosophy, AI alignment, and other concept-heavy texts.","description":"Tests whether model judgments rank argumentative critiques in the same order as expert human ratings across philosophy, AI alignment, and other concept-heavy texts.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.andrew.cmu.edu/user/coesterh/conceptual_reasoning_benchmark.html","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_87ed6a32388e715f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/conceptual-reasoning"}],"catalogSources":[{"catalog":"benchlm","sourceId":"conceptualReasoning","url":"https://benchlm.ai/benchmarks/conceptual-reasoning","paperUrl":"https://www.andrew.cmu.edu/user/coesterh/conceptual_reasoning_benchmark.html","year":null,"fullName":"Conceptual Reasoning Benchmark","format":"Average pairwise-ranking loss against expert ratings","tasks":"224 texts and 608 within-text critique pairs","successorKey":null}],"catalogCategories":["reasoning"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_if-then-otherwise-diagnosing-conditional-b_7e2fe0a2","familyId":"bmf_fefa2a334d9b","name":"CondVLN","oneLine":"CondVLN evaluates vision-language navigation agents on 11,500 generated conditional instructions across four simulators, using standard VLN metrics plus Branch Selection Accuracy and Conditional Success Rate.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.17318","pdf":"https://arxiv.org/pdf/2608.17318","project":"https://condvln.github.io/","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce CondVLN, a scene-graph-grounded benchmark for diagnosing conditional branching in vision-language navigation.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17318"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CondVLN evaluates vision-language navigation agents on 11,500 generated conditional instructions across four simulators, using standard VLN metrics plus Branch Selection Accuracy and Conditional Success Rate.","whyItMatters":"CondVLN provides controlled diagnostic signals for conditional branching failures in VLN, showing that high success rates can mask incorrect branch execution and offering a reusable testbed for instruction following under conditions.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"63993cb4778d87a0c79b64a5bc9d3971d2e7c1d9ce718e8b6f9af315b615898a"},"motivation":"Vision-language navigation agents are often evaluated on their ability to follow route-like instructions toward a fixed goal.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is clearly named CoNDVLN, has a project page, and defines reusable tasks with branch-specific diagnostics, though artifact links beyond the project page are not supplied.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce CondVLN, a scene-graph-grounded benchmark for diagnosing conditional branching in vision-language navigation."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17318","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":55,"confidence":"Low","horizon":"7d","reason":"The benchmark addresses under-explored conditional branching in VLN with multi-simulator coverage, but limited artifact evidence lowers confidence in initial attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_confbench_a42d7b7f","familyId":"bmf_32ae0b47f61b","name":"ConfBench","oneLine":"ConfBench is a calibration-specific benchmark for key information extraction from documents. It applies 20 degradation pipelines to create 1,346 variants and over 70K entity-level evaluations, spanning the accuracy spectrum for evaluating confidence estimates of VLMs.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.01792","pdf":"https://arxiv.org/pdf/2608.01792","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01792"},"evidence":{"snippet":"We introduce ConfBench, the first calibration-specific benchmark for key information extraction (KIE), built by applying 20 controlled degradation pipelines to a diverse document set, yielding 1,346 variants and 70K+ entity-level evaluations spanning the full accuracy spectrum.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01792"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"ConfBench is a calibration-specific benchmark for key information extraction from documents. It applies 20 degradation pipelines to create 1,346 variants and over 70K entity-level evaluations, spanning the accuracy spectrum for evaluating confidence estimates of VLMs.","whyItMatters":"Document processing requires trustworthy confidence scores for routing automation vs. human review. ConfBench enables systematic study of confidence estimators and calibration methods, addressing the lack of calibration-focused benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"edd9b3b99ef0608d787b5bf639c8a47d44c6ea416927fcc3438f3972ec7d529e"},"motivation":"Intelligent document processing (IDP) with vision-language models (VLMs) hinges on confidence scores trustworthy enough to route extractions between automation and human review.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.01792","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_confidencebench_21fa8523","familyId":"bmf_b9ca48dc2fb1","name":"ConfidenceBench","oneLine":"A calibration benchmark evaluating verbalized confidence estimates in frontier LLMs using Brier scores across 200 multiple-choice questions in four categories. Scores are elicited via prompting without logits, applicable to closed and open models.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20526","pdf":"https://arxiv.org/pdf/2607.20526","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20526"},"evidence":{"snippet":"We present ConfidenceBench, a calibration benchmark that evaluates verbalized confidence estimates in 15 frontier LLMs using the Brier score, a proper scoring rule that incentivises truthful probability reporting.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20526"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A calibration benchmark evaluating verbalized confidence estimates in frontier LLMs using Brier scores across 200 multiple-choice questions in four categories. Scores are elicited via prompting without logits, applicable to closed and open models.","whyItMatters":"The benchmark addresses the need to assess model calibration separately from accuracy, which is critical for trustworthy deployment. The private nature of the questions and lack of public artifacts prevent other teams from running or inspecting the benchmark, limiting its standalone utility.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f1f68150dee752cb5c5721bba768cead5af118428559027ee8c9d7762480569b"},"motivation":"Large language models (LLMs) are increasingly deployed in settings where fluent but incorrect answers can be costly.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20526","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_conflictbench_d8fc3760","familyId":"bmf_73bdd16dbae8","name":"ConflictBench","oneLine":"ConflictBench is a benchmark with ConflictScore metric to quantify how well models acknowledge conflicting evidence in grounding documents, decomposing responses into claims and labeling them against documents.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26437","pdf":"https://arxiv.org/pdf/2606.26437","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.26437"},"evidence":{"snippet":"We develop ConflictBench, a benchmark covering diverse forms of conflicts such as ambiguity, contradiction, and divergent opinions, to systematically evaluate our metric.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26437"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"ConflictBench is a benchmark with ConflictScore metric to quantify how well models acknowledge conflicting evidence in grounding documents, decomposing responses into claims and labeling them against documents.","whyItMatters":"Existing factuality metrics ignore coexisting contradictions; ConflictScore provides a nuanced measure over ConflictBench, offering a corrective feedback mechanism for improving truthfulness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"699c153bdab56cf6cd71b2fe4143d3585cbbeaaca8334edfa50d2e2e55c5aa86"},"motivation":"Existing metrics for factuality and faithfulness evaluate whether an answer is supported or contradicted by its grounding documents, but they fail to capture when both supporting and contradicting evidence coexist.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.26437","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_conlangbench_b9af66c3","familyId":"bmf_fd59187772ea","name":"ConlangBench","oneLine":"ConlangBench evaluates large language models on translation and vocabulary learning across 21 constructed languages, with a corpus of over 21 million conlang-English parallel sentence pairs and 321K vocabulary entries. The benchmark includes bidirectional translation tasks and learning-curve analysis for models trained on eight conlangs with sufficient parallel data.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.03505","pdf":"https://arxiv.org/pdf/2608.03505","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03505"},"evidence":{"snippet":"We present ConlangBench, the first large-scale benchmark for evaluating and training LLMs on 21 existing conlangs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03505"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ConlangBench evaluates large language models on translation and vocabulary learning across 21 constructed languages, with a corpus of over 21 million conlang-English parallel sentence pairs and 321K vocabulary entries. The benchmark includes bidirectional translation tasks and learning-curve analysis for models trained on eight conlangs with sufficient parallel data.","whyItMatters":"ConlangBench addresses the gap in evaluating LLMs on low-resource languages with diverse linguistic structures. It provides a controlled testbed for studying language acquisition and cross-lingual transfer, offering practical insights for model developers targeting underrepresented languages.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cc778933dc9f0238630f3713d79065c4088d322a404a6dd081accf04595e492c"},"motivation":"Constructed languages (conlangs) are intentionally created human languages with a rich tradition of linguistic creativity.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03505","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ConlangBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.03505","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_9064fade0bec0331","familyId":"catalog_family_9064fade0bec0331","name":"Connectors","oneLine":"Connectors is an OpenAI internal production benchmark measuring reliable use of connector-based tools in agentic workflows, reported as a pass rate.","description":"Connectors is an OpenAI internal production benchmark measuring reliable use of connector-based tools in agentic workflows, reported as a pass rate.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/openai-connectors","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9064fade0bec0331"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/openai-connectors"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"openai-connectors","url":"https://llm-stats.com/benchmarks/openai-connectors","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","tool calling"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_contactworld_92ad6aa1","familyId":"bmf_0a6612dbc958","name":"ContactWorld","oneLine":"ContactWorld is a benchmark and empirical study for vision-tactile world models in contact-rich manipulation, spanning 12 tasks including insertion and screwing, evaluating planning success rates.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13877","pdf":"https://arxiv.org/pdf/2606.13877","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.13877"},"evidence":{"snippet":"In this paper, we present ContactWorld, a benchmark and systematic empirical study of vision-tactile world models spanning 12 contact-rich manipulation tasks, including insertion, disassembly, screwing, and exploratory interaction.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13877"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"ContactWorld is a benchmark and empirical study for vision-tactile world models in contact-rich manipulation, spanning 12 tasks including insertion and screwing, evaluating planning success rates.","whyItMatters":"Addresses the gap in understanding which representation properties support stable long-horizon planning in contact-rich settings, offering practical guidance for multimodal model design.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"61ea2b88097ea2e06dd79d50d405a9177767e5ff1c2e893c2d32f1d945e3cfc4"},"motivation":"Contact-rich manipulation requires world models to reason over complex contact dynamics from multimodal sensory observations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13877","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_contextecho_fd140a2e","familyId":"bmf_2ceaf1c7c690","name":"ContextEcho","oneLine":"ContextEcho measures persona drift in long agentic-coding sessions, combining a 25-probe identity suite, snapshot-then-probe protocol, and three anonymized Claude Code sessions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24279","pdf":"https://arxiv.org/pdf/2605.24279","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24279"},"evidence":{"snippet":"We introduce ContextEcho, a benchmark and reusable harness for measuring persona drift at deployment scale.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24279"},"ranking":{},"description":"ContextEcho measures persona drift in long agentic-coding sessions, combining a 25-probe identity suite, snapshot-then-probe protocol, and three anonymized Claude Code sessions.","whyItMatters":"Persona drift can affect user trust and model reliability in real-world deployment, but existing evaluations may miss it. ContextEcho provides a framework for auditing persona consistency across long sessions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a75d6c933a3fd86bd10ac632ca7cf3fd7d7d292c79a2627f23d661968dbb2a42"},"motivation":"A frontier language model's acknowledged \"helpful programming assistant\" persona does not survive long agentic-coding sessions in the deployment regime that production products actually run.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24279","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_contextshift_bf97ab71","familyId":"bmf_91aae91c3f82","name":"ContextShift","oneLine":"ContextShift is a controlled benchmark that manipulates object-context relationships in COCO images to isolate context as an independent variable for object detection evaluation.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09495","pdf":"https://arxiv.org/pdf/2606.09495","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.09495"},"evidence":{"snippet":"We introduce ContextShift, a controlled benchmark that systematically manipulates object--context relationships while preserving object appearance.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09495"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ContextShift is a controlled benchmark that manipulates object-context relationships in COCO images to isolate context as an independent variable for object detection evaluation.","whyItMatters":"It reveals that standard aggregate metrics like AP can mask substantial recall loss and changes in prediction dynamics under context variation, aiding in understanding detector robustness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"311d196a3ae562287f829cd97f4a4b3e0fe207e85f2ed8750375192431ab34b9"},"motivation":"Modern object detectors achieve strong performance on standard benchmarks, yet their robustness to contextual variation remains insufficiently understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09495","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_contextweave_c388d0b9","familyId":"bmf_a571d6806b2e","name":"ContextWeave","oneLine":"ContextWeave is a longitudinal benchmark for evaluating memory in office workflows, with 1,005 executable tasks from 14 participants. It measures workspace quality and preference alignment.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04830","pdf":"https://arxiv.org/pdf/2608.04830","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.04830"},"evidence":{"snippet":"We introduce ContextWeave, a longitudinal benchmark that evaluates whether recalled experience improves downstream agent performance in realistic office-work streams.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04830"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ContextWeave is a longitudinal benchmark for evaluating memory in office workflows, with 1,005 executable tasks from 14 participants. It measures workspace quality and preference alignment.","whyItMatters":"It addresses evaluation of memory in long-horizon agent workflows, but the lack of public artifacts and scoring details makes it non-reusable without further information.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"696b147cae0380d7a7acd8296bd7539985e234930bdd8b98236ea664f191fc21"},"motivation":"Memory is essential as language agents move from isolated tasks to long-horizon, stateful workflows, yet existing evaluations often reduce it to retrieval or question answering.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04830","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_continual-learning-bench_09825e20","familyId":"bmf_427801ebd816","name":"Continual Learning Bench","oneLine":"Evaluates continual learning in LLM-based systems across six expert-validated domains with stateful tasks sharing learnable latent structure.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05661","pdf":"https://arxiv.org/pdf/2606.05661","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05661"},"evidence":{"snippet":"We introduce Continual Learning Bench (CL-Bench), the first difficult, expert-validated benchmark designed to measure whether LLM-based systems genuinely improve with experience.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05661"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates continual learning in LLM-based systems across six expert-validated domains with stateful tasks sharing learnable latent structure.","whyItMatters":"Could address the absence of high-quality benchmarks for genuine continual learning, but currently lacks verifiable public artifacts or scoring details.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e20dcf6c1741424cdc7388d88ad9d277ae2ff4388ff3fa11d9e27ee9ebdeb8d3"},"motivation":"Continual learning, the ability of AI systems to improve through sequential experience, has attracted substantial interest, but no high-quality benchmark exists to evaluate it.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05661","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_continuitybench_4edb780a","familyId":"bmf_74fd5bbd7858","name":"ContinuityBench","oneLine":"ContinuityBench evaluates stateful failover in multi-provider LLM routing. It measures Continuity Preservation Rate (CPR) and Continuity Latency Overhead (CLO) using synthetic conversation graphs under injected provider failures, with an LLM-as-a-judge scoring context preservation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-17","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.15899","pdf":"https://arxiv.org/pdf/2607.15899","project":null,"code":"https://github.com/Vishal-sys-code/continuity-bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.15899"},"evidence":{"snippet":"Furthermore, we release continuity-bench, https://github.com/Vishal-sys-code/continuity-bench, an open evaluation harness designed to stress-test context preservation under high-concurrency provider failure conditions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.15899"},"ranking":{"90d":{"score":17,"rank":393,"coverage":0.7,"confidence":"Medium"}},"description":"ContinuityBench evaluates stateful failover in multi-provider LLM routing. It measures Continuity Preservation Rate (CPR) and Continuity Latency Overhead (CLO) using synthetic conversation graphs under injected provider failures, with an LLM-as-a-judge scoring context preservation.","whyItMatters":"Production LLM deployments rely on failover mechanisms that often lose conversational context, degrading user experience. This benchmark provides a standardized way to quantify and compare context preservation and latency trade-offs across different routing architectures.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e3ed0c74b4f0d991c7b651bbde0bf14acbce3836bf9e7b5e6ffd6db48b55bb9d"},"motivation":"In production large language model (LLM) deployments, high API availability guarantees do not equate to conversational continuity.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.15899","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_continuousbench_3f96ea58","familyId":"bmf_d6ffcf4f16b1","name":"ContinuousBench","oneLine":"ContinuousBench evaluates differentially private synthetic text by measuring capability gain on QA sets derived from a new quarterly corpus (Geminon procedural data or News articles), with standardized training and evaluation harness.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01849","pdf":"https://arxiv.org/pdf/2606.01849","project":"https://peihanliu.com/posts/continuousbench.html","code":"https://github.com/plau666/ContinuousBenchEval","data":null,"hfPaper":"https://huggingface.co/papers/2606.01849"},"evidence":{"snippet":"Thus, we introduce ContinuousBench, a continuously and automatically-regenerated benchmark that measures capability gain from DP synthetic text.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01849"},"ranking":{},"description":"ContinuousBench evaluates differentially private synthetic text by measuring capability gain on QA sets derived from a new quarterly corpus (Geminon procedural data or News articles), with standardized training and evaluation harness.","whyItMatters":"Addresses the gap where existing benchmarks are nearly solvable without corpus access, providing a continuously regenerated test to determine whether DP synthesis transmits genuinely new knowledge and capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"113e95b484f3adde67025ea460c049ad054b99d2741702d34dc1c2b12308965c"},"motivation":"Differentially private (DP) text synthesis promises to unlock sensitive corpora for model training, but it remains unclear whether DP synthetic data transmits genuinely new knowledge and capabilities present only in those corpora.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01849","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ContinuousBench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/plau666/ContinuousBenchEval","role":"benchmark-publisher"},{"name":"Peihan Liu","organizationType":"academic-lab","sourceUrl":"https://peihanliu.com/posts/continuousbench.html","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e48bc81d43c698f2","familyId":"catalog_family_e48bc81d43c698f2","name":"ContPhy","oneLine":"ContPhy is a continuum physical-reasoning benchmark evaluating understanding of physical dynamics in video.","description":"ContPhy is a continuum physical-reasoning benchmark evaluating understanding of physical dynamics in video.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/contphy","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e48bc81d43c698f2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/contphy"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"contphy","url":"https://llm-stats.com/benchmarks/contphy","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","video","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_contractscrub_bad94249","familyId":"bmf_0d037e92ca7f","name":"ContractScrub","oneLine":"Evaluates LLMs on legal contract scrubbing tasks using hand-crafted contracts covering error categories like defined term misuse and inconsistent language.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.20204","pdf":"https://arxiv.org/pdf/2608.20204","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce ContractScrub, the first benchmark designed to evaluate contract scrubbing capabilities, comprising contracts hand-crafted by experienced lawyers over diverse error categories such as misuse of defined terms, incorrect references, and inconsistent language.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.20204"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLMs on legal contract scrubbing tasks using hand-crafted contracts covering error categories like defined term misuse and inconsistent language.","whyItMatters":"Provides the first formal evaluation of contract scrubbing, revealing that frontier models perform surprisingly poorly on this specific legal review task despite strong general benchmarks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"629e15406551a636fd2951b8ab72bbca614349821e0b32d5bce32d9feb4aca4f"},"motivation":"Legal work, with its heavy reliance on processing large amounts of text, is often considered one of the domains most exposed to the use of LLMs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"The paper introduces a specific evaluation object with defined error categories and a clear scoring metric (macro average recall), and the arxiv report serves as a public reference.","canonicalNameSource":"paper_title","canonicalNameEvidence":"ContractScrub: A benchmark for final review of legal contracts"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.20204","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:50.392548Z"},"attentionForecast":{"score":58,"confidence":"Medium","horizon":"7d","reason":"The domain-specific legal focus and the clear performance gap reported may attract attention from both legal AI and LLM evaluation communities."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_convbench_5f241ad0","familyId":"bmf_0c7864d7753a","name":"ConVBench","oneLine":"ConVBench is a vision-centric reasoning benchmark where each image is paired with two logically equivalent questions across six categories (action/state, complex counting, spatial reasoning, causal/intent, commonsense, temporal perception), with metrics for logical consistency and robust accuracy.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.21722","pdf":"https://arxiv.org/pdf/2607.21722","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.21722"},"evidence":{"snippet":"We introduce ConVBench, a complex vision-centric reasoning benchmark in which each image is paired with two logically equivalent questions across six categories: action and state, complex counting, spatial reasoning, causal and intent understanding, commonsense reasoning, and temporal perception.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.21722"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ConVBench is a vision-centric reasoning benchmark where each image is paired with two logically equivalent questions across six categories (action/state, complex counting, spatial reasoning, causal/intent, commonsense, temporal perception), with metrics for logical consistency and robust accuracy.","whyItMatters":"Reliable visual reasoning requires not just correctness but consistency; this benchmark measures both, offering a more rigorous test for LVLMs and motivating consistency-aware training methods.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f9c422bc1dad18c37d210b81c72bd8481f28d5f63eadc4719cd77b3151f6f9d7"},"motivation":"While Large Vision-Language Models (LVLMs) exhibit strong perceptual capabilities, they remain vulnerable in visual reasoning tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.21722","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ConVBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.21722","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_cf-yolo-context-aware-feature-refinement-f_82f82c68","familyId":"bmf_c964d28afa07","name":"Copper Tube Defect Dataset","oneLine":"Evaluates object detection of micro-defects on copper tube surfaces using 1,847 images and 4,898 bounding box instances across defect types in industrial inspection.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.28070","pdf":"https://arxiv.org/pdf/2608.28070","project":null,"code":"https://github.com/Yu-Xinda/CFYOLO-Context-Aware-Feature-Refinement-for-Camouflaged-Industrial-Micro-Defect-Detection","data":null,"hfPaper":null},"evidence":{"snippet":"To support research in this domain, we introduce the Copper Tube Defect Dataset (CTDD), a manually annotated benchmark containing 1,847 images and 4,898 boundingbox defect instances from copper-tube inspection scenarios.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.28070"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates object detection of micro-defects on copper tube surfaces using 1,847 images and 4,898 bounding box instances across defect types in industrial inspection.","whyItMatters":"Provides a labeled dataset for benchmarking detection of minute, camouflaged industrial defects, supporting comparison and development of inspection models on real-world scenarios.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T04:51:30.440525Z","inputHash":"a232b31ac463cf316c648eb735353e24243ff37f9eeb9c07b83d85de9ec6e6c0"},"motivation":"Automated detection of surface micro-defects on industrial components, such as copper tubes, is critically important for quality assurance but remains challenging due to the minute scale of anomalies and their visual camouflage against complex backgrounds.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T04:51:30.440525Z","model":"deepseek-v4-pro","decisionReason":"The abstract formally names the benchmark, describes the annotation scale, and the linked repository offers a public reference implementation; evaluation uses standard mAP and Precision metrics.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce the Copper Tube Defect Dataset (CTDD), a manually annotated benchmark containing 1,847 images and 4,898 boundingbox defect instances"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.28070","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T01:03:30.163531Z"},"attentionForecast":{"score":35,"confidence":"Low","horizon":"7d","reason":"The benchmark addresses a niche industrial defect detection task with moderate dataset size and a single-domain focus, which limits broad appeal despite the concrete release."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_copyright-bench_ed85deec","familyId":"bmf_b961b5acbe95","name":"Copyright-Bench","oneLine":"Copyright-Bench evaluates LLM agents' compliance with copyright law through realistic commercial tasks—website development, merchandise design, and pitch deck production—where agents choose between public-domain and copyrighted content, with prompt variations and time pressure.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.21799","pdf":"https://arxiv.org/pdf/2607.21799","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.21799"},"evidence":{"snippet":"To that end, we introduce Copyright-Bench, a benchmark designed to evaluate LLM agents' compliance with copyright law.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.21799"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Copyright-Bench evaluates LLM agents' compliance with copyright law through realistic commercial tasks—website development, merchandise design, and pitch deck production—where agents choose between public-domain and copyrighted content, with prompt variations and time pressure.","whyItMatters":"As agents perform commercial tasks, legal compliance is critical; this benchmark provides a structured way to assess whether agents select copyrighted materials appropriately, informing safety and regulatory considerations.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b72fd4bfbfb60c408f947409917b855deeae0c9bee1eafde3d3de1cb7ba7c860"},"motivation":"Large language model (LLM) agents increasingly perform commercial tasks that involve retrieving external content, such as images, and, where appropriate, reproducing that content.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.21799","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Copyright-Bench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.21799","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_9bad7262eb090cb1","familyId":"catalog_family_9bad7262eb090cb1","name":"CorpFin v2","oneLine":"Vals AI private benchmark for understanding long-context credit agreements.","description":"Vals AI private benchmark for understanding long-context credit agreements.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/corp_fin_v2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9bad7262eb090cb1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valscorpfinv2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsCorpFinV2","url":"https://benchlm.ai/benchmarks/valscorpfinv2","paperUrl":"https://www.vals.ai/benchmarks/corp_fin_v2","year":"2026","fullName":"Vals CorpFin v2","format":"Accuracy score","tasks":"Credit-agreement understanding tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_corporatebench_6a07b031","familyId":"bmf_dde41338c2b3","name":"CorporateBench","oneLine":"LLMs are increasingly able to answer complex questions about enterprise-scale document collections.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.27391","pdf":"https://arxiv.org/pdf/2608.27391","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.27391"},"evidence":{"snippet":"We present CorporateBench (CB), a human-validated multi-task Q&A benchmark whose scale approaches the conditions LLMs encounter in corporate communication networks, with evaluation corpora surpassing 230,000 documents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27391"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"LLMs are increasingly able to answer complex questions about enterprise-scale document collections.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"acceptance_claimed","venue":"EMNLP Findings","evidence":"Accepted to EMNLP Findings","evidenceUrl":"https://arxiv.org/abs/2608.27391","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"EMNLP Findings","reviewStatus":"accepted","decisionRaw":"Accepted to EMNLP Findings","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.27391","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to EMNLP Findings","level":"author-claim"}]}],"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_8733ada55656cd09","familyId":"catalog_family_8733ada55656cd09","name":"CorpusQA","oneLine":"CorpusQA is a multi-document, free-form long-context question answering benchmark in which a model must retrieve and reason over information distributed across a large corpus to produce open-ended answers that are scored by an LLM judge.","description":"CorpusQA is a multi-document, free-form long-context question answering benchmark in which a model must retrieve and reason over information distributed across a large corpus to produce open-ended answers that are scored by an LLM judge.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/corpusqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8733ada55656cd09"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/corpusqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"corpusqa","url":"https://llm-stats.com/benchmarks/corpusqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_a03619ad23937b0d","familyId":"catalog_family_a03619ad23937b0d","name":"CorpusQA 1M","oneLine":"CorpusQA 1M is a long-context question answering benchmark designed to evaluate models at approximately 1 million token contexts. Models are scored on accuracy when retrieving and reasoning over information distributed across an extremely long input corpus.","description":"CorpusQA 1M is a long-context question answering benchmark designed to evaluate models at approximately 1 million token contexts. Models are scored on accuracy when retrieving and reasoning over information distributed across an extremely long input corpus.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Long Context","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a03619ad23937b0d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/corpusqa1m"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/corpusqa-1m"}],"catalogSources":[{"catalog":"benchlm","sourceId":"corpusQa1m","url":"https://benchlm.ai/benchmarks/corpusqa1m","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"CorpusQA 1M","format":"Long-context QA accuracy","tasks":"Million-token corpus question answering","successorKey":null},{"catalog":"llm-stats","sourceId":"corpusqa-1m","url":"https://llm-stats.com/benchmarks/corpusqa-1m","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","long context","general"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_de43871974cdc7c1","familyId":"catalog_family_de43871974cdc7c1","name":"CountBench","oneLine":"CountBench evaluates object counting capabilities in visual understanding.","description":"CountBench evaluates object counting capabilities in visual understanding.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Reasoning","Spatial Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_de43871974cdc7c1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/countbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/countbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"countBench","url":"https://benchlm.ai/benchmarks/countbench","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"CountBench","format":"Image-grounded counting","tasks":"Visual counting tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"countbench","url":"https://llm-stats.com/benchmarks/countbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","reasoning","spatial reasoning","vision"],"catalogModelCount":7,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_173d4c77defffb6a","familyId":"catalog_family_173d4c77defffb6a","name":"CountQA","oneLine":"CountQA is a benchmark for visual object counting and quantity reasoning over images.","description":"CountQA is a benchmark for visual object counting and quantity reasoning over images.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/countqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_173d4c77defffb6a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/countqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"countqa","url":"https://llm-stats.com/benchmarks/countqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_covebench_94152ad0","familyId":"bmf_9e4ba492c102","name":"CoVEBench","oneLine":"CoVEBench evaluates compositional video editing with 416 source videos, 626 multi-point instructions, and 9,990 checklist items, using MLLM judges and objective metrics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08415","pdf":"https://arxiv.org/pdf/2606.08415","project":null,"code":"https://github.com/NJU-LINK/CoVEBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.08415"},"evidence":{"snippet":"To address this gap, we introduce CoVEBench, a compositional video editing benchmark comprising 416 curated source videos, 626 multi-point editing instructions, and 9,990 fine-grained checklist items.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":52,"hfDailySubmittedAt":"2026-06-09T00:00:00.000Z","githubStars":19,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08415"},"ranking":{"90d":{"score":48,"rank":84,"coverage":0.7,"confidence":"Medium"}},"description":"CoVEBench evaluates compositional video editing with 416 source videos, 626 multi-point instructions, and 9,990 checklist items, using MLLM judges and objective metrics.","whyItMatters":"Realistic video editing requires handling multiple coupled edits; CoVEBench provides a diagnostic testbed to reveal failures in complex instruction compliance and preservation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"24827a5a6257e03926cf514f6bd049e46bdc32156cb1b47d4448e6f561c11219"},"motivation":"While recent text-guided video editing models excel at elementary tasks (e.g., style transfer, object insertion), real-world user requests are highly compositional.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08415","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"NJU-LINK Team","organizationType":"academic-lab","sourceUrl":"https://github.com/NJU-LINK/CoVEBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_1b29385c15b421f3","familyId":"catalog_family_1b29385c15b421f3","name":"CoVoST2","oneLine":"CoVoST 2 is a large-scale multilingual speech translation corpus derived from Common Voice, covering translations from 21 languages into English and from English into 15 languages. The dataset contains 2,880 hours of speech with 78K speakers for speech translation research.","description":"CoVoST 2 is a large-scale multilingual speech translation corpus derived from Common Voice, covering translations from 21 languages into English and from English into 15 languages. The dataset contains 2,880 hours of speech with 78K speakers for speech translation research.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Speech To Text","Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/covost2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1b29385c15b421f3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/covost2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"covost2","url":"https://llm-stats.com/benchmarks/covost2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","speech to text","audio"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_aec207bcc15d0ca9","familyId":"catalog_family_aec207bcc15d0ca9","name":"CoVoST2 en-zh","oneLine":"CoVoST 2 English-to-Chinese subset is part of the large-scale multilingual speech translation corpus derived from Common Voice. This subset focuses specifically on English to Chinese speech translation tasks within the broader CoVoST 2 dataset.","description":"CoVoST 2 English-to-Chinese subset is part of the large-scale multilingual speech translation corpus derived from Common Voice. This subset focuses specifically on English to Chinese speech translation tasks within the broader CoVoST 2 dataset.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Speech To Text","Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/covost2-en-zh","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_aec207bcc15d0ca9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/covost2-en-zh"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"covost2-en-zh","url":"https://llm-stats.com/benchmarks/covost2-en-zh","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","speech to text","audio"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_34983b29734ad7b0","familyId":"catalog_family_34983b29734ad7b0","name":"CoWorkBench","oneLine":"CoWorkBench is Qwen's internal cowork benchmark for evaluating long-horizon office and productivity agent tasks across domains such as computer science, finance, law, and medicine.","description":"CoWorkBench is Qwen's internal cowork benchmark for evaluating long-horizon office and productivity agent tasks across domains such as computer science, finance, law, and medicine.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Productivity","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.8","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_34983b29734ad7b0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/coworkbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/coworkbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"coworkBench","url":"https://benchlm.ai/benchmarks/coworkbench","paperUrl":"https://qwen.ai/blog?id=qwen3.8","year":"2026","fullName":"CoWorkBench","format":"Provider-run agent score","tasks":"Long-horizon professional workflows","successorKey":null},{"catalog":"llm-stats","sourceId":"coworkbench","url":"https://llm-stats.com/benchmarks/coworkbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","productivity","reasoning","agents"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_cpi-bench_2cecc896","familyId":"bmf_6f391349d28c","name":"CPI-Bench","oneLine":"CPI-Bench evaluates image editing models across general, practical, and intelligent tasks with VLM-as-Judge scoring, including multi-image and reasoning-based editing.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.14546","pdf":"https://arxiv.org/pdf/2608.14546","project":null,"code":"https://github.com/zqyzzz/CPI-benchmark","data":null,"hfPaper":"https://huggingface.co/papers/2608.14546"},"evidence":{"snippet":"To address these limitations, we propose CPI-Bench, a Comprehensive, Practical and Intelligent benchmark for real-world image editing.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":15,"hfDailySubmittedAt":"2026-08-17T00:00:00.000Z","githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14546"},"ranking":{"30d":{"score":38,"rank":55,"coverage":0.85,"confidence":"High"},"90d":{"score":37,"rank":177,"coverage":0.7,"confidence":"Medium"}},"description":"CPI-Bench evaluates image editing models across general, practical, and intelligent tasks with VLM-as-Judge scoring, including multi-image and reasoning-based editing.","whyItMatters":"It provides a comprehensive benchmark that captures real-world deployment scenarios and reasoning demands, with scoring aligned to human preferences for reliable model comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0604b1a0b8824409d44a66b794d09d481d64bc8fb557dd3a42663a95ffe1b6e3"},"motivation":"With the rapid advancement of image editing models and their widespread application across various domains, there is an increasingly urgent need to deploy these model capabilities directly into real-world scenarios.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14546","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TaobaoTmall-AlgorithmProducts","organizationType":"company-research-lab","sourceUrl":"https://github.com/zqyzzz/CPI-benchmark","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_cra-bench_d9f4029f","familyId":"bmf_6c4b07513834","name":"CRA-Bench","oneLine":"A session-layer framework for tracking conversational risk accumulation in multi-turn LLM systems, including CRA-Bench datasets and trajectory-native evaluation protocols.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19361","pdf":"https://arxiv.org/pdf/2607.19361","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19361"},"evidence":{"snippet":"To benchmark CRA, we release CRA-Bench v0.1 (1,200 eight-turn sessions across three threat families with topic-matched benign twins), CRA-Bench v0.2 (LLM-paraphrased variants to reduce template artifacts), and an extended 5-family set (2,000 sessions adding persona priming and context stuffing).","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19361"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A session-layer framework for tracking conversational risk accumulation in multi-turn LLM systems, including CRA-Bench datasets and trajectory-native evaluation protocols.","whyItMatters":"Addresses the gap in evaluating guardrails for multi-turn dialogues where benign turns compose into harm, providing session-level scoring metrics beyond isolated prompt-response checks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5fc97dec5b31864e987507e49e99a3e167e785a4fb449015936fb75a7920c402"},"motivation":"Most safety guardrails for large language models (LLMs) evaluate each prompt-response pair in isolation, which misses failures that arise only over a dialogue as benign turns compose into harm.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19361","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_crab-bench_a7bbb55e","familyId":"bmf_5a353a7a7b36","name":"CRAB-Bench","oneLine":"CRAB-Bench evaluates LLM agents on tasks generated via a constraint graph over multiple interdependent entities, using the RUSE user simulator with imperfect behavior. Scoring is pass@1 against valid solutions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01815","pdf":"https://arxiv.org/pdf/2606.01815","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01815"},"evidence":{"snippet":"We introduce CRAB-Bench (Constraint-based Realistic Agent Benchmark) and RUSE (Realistic User Simulation Engine) to address this gap.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01815"},"ranking":{},"description":"CRAB-Bench evaluates LLM agents on tasks generated via a constraint graph over multiple interdependent entities, using the RUSE user simulator with imperfect behavior. Scoring is pass@1 against valid solutions.","whyItMatters":"Establishes an evaluation setting for agent performance under complex task dependencies and simulated realistic user behavior, measuring task-solving ability and conversational quality in service scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b466f9c6d77fa5219d3edbe8f546d6698546f0636cccd8a51e3b645b4fb9c869"},"motivation":"Evaluating LLM agents in realistic service scenarios requires complex task dependencies, imperfect user behavior, and an evaluation that accommodates multiple valid solutions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01815","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_crackedpdfs_1a52f043","familyId":"bmf_e110c4f10728","name":"CrackedPDFs","oneLine":"Evaluates hidden prompt injection detection in PDFs through classification and paired ranking tasks, using 29,322 generated PDFs from 4,983 base documents. Includes frozen splits, features, and metrics for reproducibility.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19396","pdf":"https://arxiv.org/pdf/2607.19396","project":"https://doi.org/10.5281/zenodo.21735803","code":"https://github.com/volkthienpreecha/crackedpdfs/releases/tag/v1.0.0-paper","data":"https://huggingface.co/datasets/volkthienpreecha/crackedpdfs","hfPaper":"https://huggingface.co/papers/2607.19396"},"evidence":{"snippet":"We introduce CrackedPDFs, a controlled benchmark for hidden prompt injection in PDFs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":269,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2607.19396"},"ranking":{"90d":{"score":19,"rank":390,"coverage":1.0,"confidence":"High","datasetDownloadRank":33,"datasetRankPopulation":66}},"description":"Evaluates hidden prompt injection detection in PDFs through classification and paired ranking tasks, using 29,322 generated PDFs from 4,983 base documents. Includes frozen splits, features, and metrics for reproducibility.","whyItMatters":"Addresses the gap in evaluating defenses against prompt injections embedded in PDF structure, where flattening can hide malicious instructions. Provides a controlled paired benchmark with confounding controls to assess whether detectors generalize beyond superficial cues.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b5aa65e925fc117358b2a0be920f582376b262ebdf182646d53138ebac700c78"},"motivation":"Document-based LLM systems often flatten a PDF before guardrails inspect it.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19396","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_craftbench_a8850dcb","familyId":"bmf_3195c5732e34","name":"CraftBench","oneLine":"CraftBench evaluates scientific figure generation across three figure types and four input conditions, with human-drawn targets and a referenced VLM judge for scoring.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30611","pdf":"https://arxiv.org/pdf/2605.30611","project":null,"code":"https://github.com/HaozheZhao/Crafter","data":null,"hfPaper":"https://huggingface.co/papers/2605.30611"},"evidence":{"snippet":"Moreover, we introduce CraftBench, a benchmark spanning three figure types and four input conditions with human quality annotation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":253,"hfDailySubmittedAt":null,"githubStars":156,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30611"},"ranking":{},"description":"CraftBench evaluates scientific figure generation across three figure types and four input conditions, with human-drawn targets and a referenced VLM judge for scoring.","whyItMatters":"Automated figure generation lacks comprehensive benchmarks covering diverse types and conditions. CraftBench provides a standardized evaluation to measure progress in editable scientific figure generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"952c6bd1ea2c9296a9ea9387df8a021df3395a69d8cf06337df7ad06d4d18651"},"motivation":"Scientific figures are among the most effective means of communicating complex research ideas, yet producing publication-quality illustrations remains one of the most labor-intensive parts of paper preparation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30611","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HaozheZhao/Crafter","organizationType":"community","sourceUrl":"https://github.com/HaozheZhao/Crafter","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_53852581fcc7cfc2","familyId":"catalog_family_53852581fcc7cfc2","name":"CRAG","oneLine":"CRAG (Comprehensive RAG Benchmark) is a factual question answering benchmark consisting of 4,409 question-answer pairs across 5 domains (finance, sports, music, movie, open domain) and 8 question categories. The benchmark includes mock APIs to simulate web and Knowledge Graph search, designed to represent the diverse and dynamic nature of real-world QA tasks with temporal dynamism ranging from years to seconds. It evaluates retrieval-augmented generation systems for trustworthy question answering.","description":"CRAG (Comprehensive RAG Benchmark) is a factual question answering benchmark consisting of 4,409 question-answer pairs across 5 domains (finance, sports, music, movie, open domain) and 8 question categories. The benchmark includes mock APIs to simulate web and Knowledge Graph search, designed to represent the diverse and dynamic nature of real-world QA tasks with temporal dynamism ranging from years to seconds. It evaluates retrieval-augmented generation systems for trustworthy question answering.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Reasoning","Search","Finance","Economics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/crag","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_53852581fcc7cfc2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/crag"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"crag","url":"https://llm-stats.com/benchmarks/crag","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","search","finance","economics"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"specific"},{"id":"bm_crag-mm-diagnostics_8195c595","familyId":"bmf_572d27c0d445","name":"CRAG-MM-Diagnostics","oneLine":"CRAG-MM-Diagnostics is a diagnostic benchmark for knowledge-intensive visual question answering (KI-VQA) with stage-wise annotations to isolate visual grounding, object identification, and knowledge retrieval/reasoning, including metadata like target ROIs and visual complexity scores.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Information retrieval","Factuality"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.21155","pdf":"https://arxiv.org/pdf/2607.21155","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.21155"},"evidence":{"snippet":"To analyze the full KI-VQA pipeline, we introduce CRAG-MM-Diagnostics, a diagnostic benchmark with stage-wise data annotations that isolate 1) language-based visual grounding, 2) object identification, and 3) knowledge retrieval and reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.21155"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CRAG-MM-Diagnostics is a diagnostic benchmark for knowledge-intensive visual question answering (KI-VQA) with stage-wise annotations to isolate visual grounding, object identification, and knowledge retrieval/reasoning, including metadata like target ROIs and visual complexity scores.","whyItMatters":"End-task accuracy alone obscures failure sources in KI-VQA; this benchmark enables stage-wise analysis to identify bottlenecks, guiding improvements in multimodal retrieval-augmented generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"992d635f361217d634d58c3a48155ba4b595996d3322d10dcf5c4d9e9cf7b0d3"},"motivation":"Knowledge-Intensive Visual Question Answering (KI-VQA) benchmarks evaluate Vision-Language Models (VLMs) as multimodal knowledge assistants by requiring external information beyond a provided image to answer questions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026","evidence":"Accepted to ECCV 2026","evidenceUrl":"https://arxiv.org/abs/2607.21155","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026","reviewStatus":"accepted","decisionRaw":"Accepted to ECCV 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.21155","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to ECCV 2026","level":"author-claim"}]}],"publishers":[{"name":"CRAG-MM-Diagnostics Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.21155","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_05a6cb704e35a0f8","familyId":"catalog_family_05a6cb704e35a0f8","name":"Creative Writing v3","oneLine":"EQ-Bench Creative Writing v3 is an LLM-judged creative writing benchmark that evaluates models across 32 writing prompts with 3 iterations per prompt. Uses a hybrid scoring system combining rubric assessment and Elo ratings through pairwise comparisons. Challenges models in areas like humor, romance, spatial awareness, and unique perspectives to assess emotional intelligence and creative writing capabilities.","description":"EQ-Bench Creative Writing v3 is an LLM-judged creative writing benchmark that evaluates models across 32 writing prompts with 3 iterations per prompt. Uses a hybrid scoring system combining rubric assessment and Elo ratings through pairwise comparisons. Challenges models in areas like humor, romance, spatial awareness, and unique perspectives to assess emotional intelligence and creative writing capabilities.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Creativity","Writing"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/creative-writing-v3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_05a6cb704e35a0f8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/creative-writing-v3"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"creative-writing-v3","url":"https://llm-stats.com/benchmarks/creative-writing-v3","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["creativity","writing"],"catalogModelCount":13,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d7cac9c36bec7188","familyId":"catalog_family_d7cac9c36bec7188","name":"CreativeWork","oneLine":"CreativeWork evaluates agents on open-ended creative production tasks within realistic tool and application environments, measuring the quality and completeness of generated deliverables.","description":"CreativeWork evaluates agents on open-ended creative production tasks within realistic tool and application environments, measuring the quality and completeness of generated deliverables.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/creativework","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d7cac9c36bec7188"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/creativework"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"creativework","url":"https://llm-stats.com/benchmarks/creativework","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_criticalscm-bench_de361f05","familyId":"bmf_4b271b812478","name":"CriticalSCM-Bench","oneLine":"CriticalSCM-Bench v1 is a synthetic supply-chain benchmark with causal ground truth, paired factual/counterfactual rollouts, and an explicit net-value objective. It evaluates intervention ranking policies across semiconductor, critical-material, and digital-infrastructure archetypes under partial and delayed information, with stress tests on fidelity, timing, cost, and held-out disruptions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.LG"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.11154","pdf":"https://arxiv.org/pdf/2608.11154","project":null,"code":"https://github.com/dyshang/dacri-criticalscm-bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.11154"},"evidence":{"snippet":"We present CriticalSCM-Bench v1, a controlled synthetic benchmark with causal ground truth, paired factual/counterfactual rollouts, and an explicit net-value objective.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11154"},"ranking":{"30d":{"score":23,"rank":154,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":358,"coverage":0.55,"confidence":"Low"}},"description":"CriticalSCM-Bench v1 is a synthetic supply-chain benchmark with causal ground truth, paired factual/counterfactual rollouts, and an explicit net-value objective. It evaluates intervention ranking policies across semiconductor, critical-material, and digital-infrastructure archetypes under partial and delayed information, with stress tests on fidelity, timing, cost, and held-out disruptions.","whyItMatters":"The benchmark addresses the gap between detection and decision by quantifying recoverable net value of interventions, helping choose between adaptive ranking and simpler structural policies based on domain conditions and out-of-distribution robustness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d0f1674313ce4334d618dc923948904ab6122956d8f3cd02b28d0e09f79ae244"},"motivation":"Detecting or attributing a supply-chain disruption is not the same as selecting the intervention that maximizes recoverable net value.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"presentation at the International Conference on Electrical, Computer, Communications and Mechatronics Engineering (ICECC","evidence":"Accepted for presentation at the International Conference on Electrical, Computer, Communications and Mechatronics Engineering (ICECCME 2026), 15--17 October 2026, Bali, Indonesia. 5 tables; no figures. Benchmark and code: https://github.com/dyshang/dacri-criticalscm-bench","evidenceUrl":"https://arxiv.org/abs/2608.11154","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"presentation at the International Conference on Electrical, Computer, Communications and Mechatronics Engineering (ICECC","reviewStatus":"accepted","decisionRaw":"Accepted for presentation at the International Conference on Electrical, Computer, Communications and Mechatronics Engineering (ICECCME 2026), 15--17 October 2026, Bali, Indonesia. 5 tables; no figures. Benchmark and code: https://github.com/dyshang/dacri-criticalscm-bench","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.11154","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted for presentation at the International Conference on Electrical, Computer, Communications and Mechatronics Engineering (ICECCME 2026), 15--17 October 2026, Bali, Indonesia. 5 tables; no figures. Benchmark and code: https://github.com/dyshang/dacri-criticalscm-bench","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_6093c843839ad76f","familyId":"catalog_family_6093c843839ad76f","name":"CritPt","oneLine":"CritPT is a challenging reasoning benchmark reported by Qwen for evaluating frontier mathematical and critical problem-solving capability.","description":"CritPT is a challenging reasoning benchmark reported by Qwen for evaluating frontier mathematical and critical problem-solving capability.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/evaluations/critpt","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6093c843839ad76f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/critpt"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/critpt"}],"catalogSources":[{"catalog":"benchlm","sourceId":"critpt","url":"https://benchlm.ai/benchmarks/critpt","paperUrl":"https://artificialanalysis.ai/evaluations/critpt","year":"2026","fullName":"Critical Physics Tasks","format":"Accuracy","tasks":"Research-level physics questions","successorKey":null},{"catalog":"llm-stats","sourceId":"critpt","url":"https://llm-stats.com/benchmarks/critpt","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","math"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_cronos_36061fdb","familyId":"bmf_6526b48f897f","name":"CRONOS","oneLine":"CRONOS is an intervention-based benchmark for evaluating counterfactual physical consistency in video prediction models. It provides a photorealistic Unreal Engine environment with controlled videos of physical events (collision, occlusion, fall) while intervening on viewpoint, scene, object category, and object appearance. The evaluation protocol defines six metrics computed via video segmentation, tracking, 3D reconstruction, and VLM-based task performance.","area":"Vision & 3D","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.23699","pdf":"https://arxiv.org/pdf/2605.23699","project":null,"code":"https://github.com/GenIntel/CRONOS-benchmark","data":null,"hfPaper":"https://huggingface.co/papers/2605.23699"},"evidence":{"snippet":"We introduce CRONOS, an intervention-based benchmark designed to evaluate counterfactual physical consistency: whether a model's predictions of physical events respond appropriately to controlled changes in the visual input, such as variations of scene context, viewpoint, object appearance, and object category.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":11,"hfDailySubmittedAt":null,"githubStars":8,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.23699"},"ranking":{},"description":"CRONOS is an intervention-based benchmark for evaluating counterfactual physical consistency in video prediction models. It provides a photorealistic Unreal Engine environment with controlled videos of physical events (collision, occlusion, fall) while intervening on viewpoint, scene, object category, and object appearance. The evaluation protocol defines six metrics computed via video segmentation, tracking, 3D reconstruction, and VLM-based task performance.","whyItMatters":"Current video models often rely on superficial correlations rather than causal structure. CRONOS enables systematic diagnosis of how prediction quality degrades under controlled interventions, providing a concrete target for developing models robust to variations in viewpoint and context.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c0521a32387001d3b42c041ed4cfef544c0ae34c4b83ee919a0d26e527159a47"},"motivation":"Video prediction is increasingly viewed as a path toward generalizable world models, yet it remains unclear whether these systems learn underlying causal structure or merely exploit superficial visual correlations for future prediction.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.23699","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GenIntel","organizationType":"academic-lab","sourceUrl":"https://github.com/GenIntel/CRONOS-benchmark","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_cross-engine-humanoid-benchmark_f797b695","familyId":"bmf_0ef7a688a656","name":"Cross-Engine Humanoid Benchmark","oneLine":"Compares MuJoCo and PyBullet on a scripted squat-and-recover task for the Unitree G1 humanoid, measuring joint tracking RMSE, contact forces, CoM deviation, and realtime factor, with PSO-based calibration.","area":"Language & Knowledge","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.AI"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/tiago369/cross-engine-humanoid-benchmark","pdf":null,"project":null,"code":"https://github.com/tiago369/cross-engine-humanoid-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"cross-engine-humanoid-benchmark Cross-engine (MuJoCo vs PyBullet) humanoid benchmark with PSO-based sim calibration # Cross-Engine Humanoid Benchmark [![benchmark](https://github.com/tiago369/cross-engine-humanoid-benchmark/actions/workflows/benchmark.yml/badge.svg)](https://github.com/tiago369/cross-engine-humanoid-benchmark/actions/workflows/benchmark.yml) **Build, configure, and operate a robotics simulation environment; run a standardized benchmark against a scripted reference task; evaluate","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:tiago369/cross-engine-humanoid-benchmark"},"ranking":{"30d":{"score":23,"rank":126,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":330,"coverage":0.55,"confidence":"Low"}},"description":"Compares MuJoCo and PyBullet on a scripted squat-and-recover task for the Unitree G1 humanoid, measuring joint tracking RMSE, contact forces, CoM deviation, and realtime factor, with PSO-based calibration.","whyItMatters":"Provides a standardized protocol to quantify physics engine differences on the same robot, including a calibration method that can guide sim-to-real transfer for humanoid control.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"573ee0c03ab03ad16c89998a37512583bab88e939aa2b4593bd8d8ceb51dea58"},"motivation":"cross-engine-humanoid-benchmark Cross-engine (MuJoCo vs PyBullet) humanoid benchmark with PSO-based sim calibration # Cross-Engine Humanoid Benchmark [![benchmark](https://github.com/tiago369/cross-engine-humanoid-benchmark/actions/workflows/benchmark.yml/badge.svg)](https://github.com/tiago369/cross-engine-humanoid-benchmark/actions/workflows/benchmark.yml) **Build, configure, and operate a robotics simulation envi…","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/tiago369/cross-engine-humanoid-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":30,"confidence":"Low","horizon":"7d","reason":"The benchmark is narrowly focused on engine comparison for a specific robot and task, with limited broad appeal beyond robotics simulation practitioners."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_848579aa1776b705","familyId":"catalog_family_848579aa1776b705","name":"CrossVid","oneLine":"CrossVid evaluates cross-video reasoning, requiring models to integrate information across multiple videos.","description":"CrossVid evaluates cross-video reasoning, requiring models to integrate information across multiple videos.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/crossvid","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_848579aa1776b705"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/crossvid"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"crossvid","url":"https://llm-stats.com/benchmarks/crossvid","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","video","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_crossview_b52b71d9","familyId":"bmf_fb565242d442","name":"CrossView","oneLine":"CrossView is a multi-camera video question-answering benchmark spanning autonomous driving, security surveillance, egocentric/exocentric video, and robotics. It evaluates vision-language models on tasks that require joint reasoning across multiple simultaneous camera views, including resolving occlusions and integrating evidence across perspectives.","area":"Multimodal","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Automotive"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15539","pdf":"https://arxiv.org/pdf/2608.15539","project":"https://utaustin-swarmlab.github.io/CrossView","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.15539"},"evidence":{"snippet":"We introduce CrossView, a multi-camera video question-answering benchmark spanning autonomous driving, security surveillance, egocentric/exocentric video, and robotics.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15539"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CrossView is a multi-camera video question-answering benchmark spanning autonomous driving, security surveillance, egocentric/exocentric video, and robotics. It evaluates vision-language models on tasks that require joint reasoning across multiple simultaneous camera views, including resolving occlusions and integrating evidence across perspectives.","whyItMatters":"Existing video benchmarks focus on single-camera settings, leaving multi-camera reasoning unmeasured. CrossView provides a standardized evaluation for a capability critical to real-world applications like autonomous vehicles and surveillance, enabling comparison of models on tasks that scale with viewpoint number and require cross-view integration.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"83073cc8b29d61af4121191ab872902c2af21700d4c3fb8a7416bc1431cf7279"},"motivation":"Video understanding benchmarks have long centered on single-camera settings, where modern multi-modal language models achieve strong performance across image and video tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15539","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"UT Austin Swarm Lab","organizationType":"academic-lab","sourceUrl":"https://utaustin-swarmlab.github.io/CrossView","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_558ad5fcac1a5329","familyId":"catalog_family_558ad5fcac1a5329","name":"CRPErelation","oneLine":"Clinical reasoning problems evaluation benchmark for assessing diagnostic reasoning and medical knowledge application capabilities.","description":"Clinical reasoning problems evaluation benchmark for assessing diagnostic reasoning and medical knowledge application capabilities.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Reasoning","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/crperelation","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_558ad5fcac1a5329"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/crperelation"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"crperelation","url":"https://llm-stats.com/benchmarks/crperelation","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","healthcare"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_crs-bench_17476b11","familyId":"bmf_2bc4591a1546","name":"CRS-Bench","oneLine":"Evaluates 15 pretrained image encoder families on ISIC 2019, APTOS 2019, and CheXpert using discrimination, calibration, label efficiency, and robustness, summarized by the Clinical Reliability Score.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["eess.IV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-22","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.22059v1","pdf":"https://arxiv.org/pdf/2608.22059v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce CRS-Bench, a controlled benchmark for multi-objective medical encoder selection.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22059"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates 15 pretrained image encoder families on ISIC 2019, APTOS 2019, and CheXpert using discrimination, calibration, label efficiency, and robustness, summarized by the Clinical Reliability Score.","whyItMatters":"Moves medical encoder selection beyond AUROC by providing a multi-axis reliability profile, helping practitioners choose models robust to distribution shift and label scarcity.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"95da0854ce0885b82aa43dcc56948e744a0f097cdd237f3865388318464cc691"},"motivation":"Pretrained image encoders are central to medical image classification, where expert annotation is costly and task-specific cohorts are often limited.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The paper formally introduces CRS-Bench with a clear scoring contract and results, but no code or data link is provided, so public reuse cannot be confirmed.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce CRS-Bench, a controlled benchmark for multi-objective medical encoder selection."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22059v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses a timely gap in medical model selection and is presented in an arXiv paper with broad applicability, likely attracting moderate early attention."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_cruisebench_5f43e5a1","familyId":"bmf_3ac2a92019a7","name":"CruiseBench","oneLine":"CruiseBench evaluates remaining useful life prediction for aircraft engines using a fixed cruise-stage protocol derived from N-CMAPSS. It provides a reproducible sub-benchmark with datasets and baseline results for LSTM, GRU, TCN, and TSMixer models under specific settings.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19380","pdf":"https://arxiv.org/pdf/2607.19380","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19380"},"evidence":{"snippet":"To mitigate this issue, this paper proposes CruiseBench, a cruise-stage RUL benchmark derived from N-CMAPSS.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19380"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CruiseBench evaluates remaining useful life prediction for aircraft engines using a fixed cruise-stage protocol derived from N-CMAPSS. It provides a reproducible sub-benchmark with datasets and baseline results for LSTM, GRU, TCN, and TSMixer models under specific settings.","whyItMatters":"RUL prediction is critical for maintenance planning, but existing benchmarks like N-CMAPSS lack evaluation control due to full-flight records. CruiseBench offers a controlled, reproducible setting for fair comparison of RUL models, reducing variability from stage and preprocessing choices.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"91d7d5726156b546179187a0567da6519e6cdabefa27cbcd57a71e5a8688fce1"},"motivation":"Remaining useful life (RUL) prediction estimates how long an engine can continue safe operation and is central to maintenance planning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19380","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_08f47d035bbfbc28","familyId":"catalog_family_08f47d035bbfbc28","name":"CRUX-O","oneLine":"CRUXEval-O (output prediction) is part of the CRUXEval benchmark consisting of 800 Python functions (3-13 lines) designed to evaluate AI models' capabilities in code reasoning, understanding, and execution. The benchmark tests models' ability to predict correct function outputs given function code and inputs, focusing on short problems that a good human programmer should be able to solve in a minute.","description":"CRUXEval-O (output prediction) is part of the CRUXEval benchmark consisting of 800 Python functions (3-13 lines) designed to evaluate AI models' capabilities in code reasoning, understanding, and execution. The benchmark tests models' ability to predict correct function outputs given function code and inputs, focusing on short problems that a good human programmer should be able to solve in a minute.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/crux-o","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_08f47d035bbfbc28"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/crux-o"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"crux-o","url":"https://llm-stats.com/benchmarks/crux-o","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_181f2f5e13ac9448","familyId":"catalog_family_181f2f5e13ac9448","name":"CRUXEval-Input-CoT","oneLine":"CRUXEval input prediction task with Chain of Thought (CoT) prompting. Part of the CRUXEval benchmark for code reasoning, understanding, and execution evaluation. Given a Python function and its expected output, the task is to predict the appropriate input using chain-of-thought reasoning. Consists of 800 Python functions (3-13 lines) designed to evaluate code comprehension and reasoning capabilities.","description":"CRUXEval input prediction task with Chain of Thought (CoT) prompting. Part of the CRUXEval benchmark for code reasoning, understanding, and execution evaluation. Given a Python function and its expected output, the task is to predict the appropriate input using chain-of-thought reasoning. Consists of 800 Python functions (3-13 lines) designed to evaluate code comprehension and reasoning capabilities.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cruxeval-input-cot","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_181f2f5e13ac9448"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cruxeval-input-cot"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cruxeval-input-cot","url":"https://llm-stats.com/benchmarks/cruxeval-input-cot","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_981b30ba7456c8df","familyId":"catalog_family_981b30ba7456c8df","name":"CruxEval-O","oneLine":"CruxEval-O is the output prediction task of the CRUXEval benchmark, designed to evaluate code reasoning, understanding, and execution capabilities. It consists of 800 Python functions (3-13 lines) where models must predict the output given a function and input. The benchmark tests fundamental code execution reasoning abilities and goes beyond simple code generation to assess deeper understanding of program behavior.","description":"CruxEval-O is the output prediction task of the CRUXEval benchmark, designed to evaluate code reasoning, understanding, and execution capabilities. It consists of 800 Python functions (3-13 lines) where models must predict the output given a function and input. The benchmark tests fundamental code execution reasoning abilities and goes beyond simple code generation to assess deeper understanding of program behavior.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cruxeval-o","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_981b30ba7456c8df"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cruxeval-o"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cruxeval-o","url":"https://llm-stats.com/benchmarks/cruxeval-o","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d7bd94f44513f617","familyId":"catalog_family_d7bd94f44513f617","name":"CRUXEval-Output-CoT","oneLine":"CRUXEval-O (output prediction) with Chain-of-Thought prompting. Part of the CRUXEval benchmark consisting of 800 Python functions (3-13 lines) designed to evaluate code reasoning, understanding, and execution capabilities. The output prediction task requires models to predict the output of a given Python function with specific inputs, evaluated using chain-of-thought reasoning methodology.","description":"CRUXEval-O (output prediction) with Chain-of-Thought prompting. Part of the CRUXEval benchmark consisting of 800 Python functions (3-13 lines) designed to evaluate code reasoning, understanding, and execution capabilities. The output prediction task requires models to predict the output of a given Python function with specific inputs, evaluated using chain-of-thought reasoning methodology.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cruxeval-output-cot","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d7bd94f44513f617"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cruxeval-output-cot"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cruxeval-output-cot","url":"https://llm-stats.com/benchmarks/cruxeval-output-cot","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_crystalxrd-bench_0f28a22b","familyId":"bmf_6125c37d5187","name":"CrystalXRD-Bench","oneLine":"CrystalXRD-Bench evaluates vision-language models on XRD peak indexing, requiring the model to identify HKL indices from rendered XRD images and CIF text across 250 samples from 10 databases.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29446","pdf":"https://arxiv.org/pdf/2605.29446","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29446"},"evidence":{"snippet":"We introduce CrystalXRD-Bench, a 250-sample benchmark built from 10 public crystallographic databases for a single task: recover the full set of HKLs contributing to the highest-intensity peak in an XRD pattern.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29446"},"ranking":{},"description":"CrystalXRD-Bench evaluates vision-language models on XRD peak indexing, requiring the model to identify HKL indices from rendered XRD images and CIF text across 250 samples from 10 databases.","whyItMatters":"Existing multimodal benchmarks do not test this specialized scientific skill. CrystalXRD-Bench isolates visual extraction and crystallographic reasoning errors, providing a focused evaluation for quantitative figure understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"86a077ed9c14f9083520b22a9ecd975da848d89eeabe973fc7114440969eec9e"},"motivation":"Miller-index identification from powder XRD patterns requires capabilities untested by existing multimodal benchmarks: the model must read a narrow peak location from a rendered scientific curve and then connect that observation to multi-step crystallographic reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29446","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_c9d74c7adac2117e","familyId":"catalog_family_c9d74c7adac2117e","name":"CSimpleQA","oneLine":"Chinese SimpleQA is the first comprehensive Chinese benchmark to evaluate the factuality ability of language models to answer short questions. It contains 3,000 high-quality questions spanning 6 major topics with 99 diverse subtopics, designed to assess Chinese factual knowledge across humanities, science, engineering, culture, and society.","description":"Chinese SimpleQA is the first comprehensive Chinese benchmark to evaluate the factuality ability of language models to answer short questions. It contains 3,000 high-quality questions spanning 6 major topics with 99 diverse subtopics, designed to assess Chinese factual knowledge across humanities, science, engineering, culture, and society.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/csimpleqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c9d74c7adac2117e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/csimpleqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"csimpleqa","url":"https://llm-stats.com/benchmarks/csimpleqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","general"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_cstutorbench_cf78c812","familyId":"bmf_0e6e4bdbb6d8","name":"CSTutorBench","oneLine":"CSTutorBench evaluates small language models as tutors in VEX VR block-based programming, with 17 scenario-based questions scored via a rubric and LLM-as-judge.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Code generation"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.05571","pdf":"https://arxiv.org/pdf/2607.05571","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.05571"},"evidence":{"snippet":"We introduce CSTutorBench, a benchmark for evaluating language models as CS tutors in VEX VR, a block-based robotics environment.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.05571"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CSTutorBench evaluates small language models as tutors in VEX VR block-based programming, with 17 scenario-based questions scored via a rubric and LLM-as-judge.","whyItMatters":"The benchmark addresses the gap in evaluating SLMs for block-based programming tutoring, where models often lack domain-specific training.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d042a9cc0d83045cb479a5562f2ff6a122de902e8ed06dd8373c6364e06c7f70"},"motivation":"Large language models are increasingly explored as AI tutors, yet deploying them in K-12 settings raises concerns around privacy, cost, and reliance on proprietary models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"publication_reported","venue":"SLM4ED'26: The 1st Workshop of Small Language Models for Education (SLM4ED). AIED 2026. Seoul, Republic of Korea","evidence":"SLM4ED'26: The 1st Workshop of Small Language Models for Education (SLM4ED). AIED 2026. Seoul, Republic of Korea","evidenceUrl":"https://arxiv.org/abs/2607.05571","source":"arxiv-journal-reference","evidenceLevel":"strong-author-metadata","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publications":[{"venueName":"SLM4ED'26: The 1st Workshop of Small Language Models for Education (SLM4ED). AIED 2026. Seoul, Republic of Korea","publicationStatus":"published","evidence":[{"sourceType":"arxiv-journal-reference","sourceUrl":"https://arxiv.org/abs/2607.05571","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"SLM4ED'26: The 1st Workshop of Small Language Models for Education (SLM4ED). AIED 2026. Seoul, Republic of Korea","level":"strong-author-metadata"}]}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_csvfidelity-bench_89145848","familyId":"bmf_8cb890c5c077","name":"CSVFidelity-Bench","oneLine":"CSVFidelity-Bench evaluates knowledge graph construction from statistical tables, focusing on the effect of extraction schema and format coupling on fidelity. It includes 15 datasets, 11 Type-II and 4 Type-III tables, with 1,892 gold standard facts across 6 domains.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.21974","pdf":"https://arxiv.org/pdf/2605.21974","project":"https://anonymous.4open.science/r/sge_lightrag-BE19","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.21974"},"evidence":{"snippet":"To support fidelity-aware evaluation, we release CSVFidelity-Bench.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.21974"},"ranking":{},"description":"CSVFidelity-Bench evaluates knowledge graph construction from statistical tables, focusing on the effect of extraction schema and format coupling on fidelity. It includes 15 datasets, 11 Type-II and 4 Type-III tables, with 1,892 gold standard facts across 6 domains.","whyItMatters":"Addresses the gap in evaluating knowledge graph construction fidelity, particularly the overlooked interaction between serialization format and schema constraints, which can cause catastrophic coverage loss.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3dde2798e517b59f957714c3955f28c0ca2e59cf3ec6cd2539d2865c1f749af2"},"motivation":"An extraction schema should not reduce knowledge graph fidelity.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.21974","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ctbench_d210f8ea","familyId":"bmf_a5ad32879a12","name":"CTBench","oneLine":"Evaluates AI agents on telecom network troubleshooting tasks, focusing on root cause analysis and path restoration, using expert-grounded metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12002","pdf":"https://arxiv.org/pdf/2608.12002","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12002"},"evidence":{"snippet":"In this paper, we introduce CTBench, a public benchmark for assessing whether an agent behaves like a competent telecom troubleshooting engineer.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12002"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates AI agents on telecom network troubleshooting tasks, focusing on root cause analysis and path restoration, using expert-grounded metrics.","whyItMatters":"Fills the gap in evaluating AI agents for realistic telecom operations, emphasizing evidence-based diagnosis and practical constraints.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fad9dec77a807219b275ceb8cd3b26136f319213f2e66a328a5f67f484c238a7"},"motivation":"Agents are increasingly considered for automating network operations and maintenance, where engineers must diagnose network faults, optimize configurations to enhance services, and reduce operational costs while acting under strict constraints.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12002","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_80c525d95c1b5514","familyId":"catalog_family_80c525d95c1b5514","name":"CTI-REALM","oneLine":"A cybersecurity benchmark that measures whether an agent can turn raw threat-intelligence reports into working detection rules.","description":"A cybersecurity benchmark that measures whether an agent can turn raw threat-intelligence reports into working detection rules.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://sakana.ai/fugu-cyber-release/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_80c525d95c1b5514"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/ctirealm"}],"catalogSources":[{"catalog":"benchlm","sourceId":"ctiRealm","url":"https://benchlm.ai/benchmarks/ctirealm","paperUrl":"https://sakana.ai/fugu-cyber-release/","year":"2026","fullName":"CTI-REALM","format":"Success rate","tasks":"Threat-intelligence-to-detection-rule workflows","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_cultural-moment-benchmark_aacba1dd","familyId":"bmf_6b3372967557","name":"Cultural Moment Benchmark","oneLine":"Evaluates video cultural reasoning and grounding on 306 expert-curated concepts from Southeast Asia, scoring naming, visual recognition, and temporal localization.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":0.95,"links":{"report":"http://arxiv.org/abs/2608.23065v1","pdf":"https://arxiv.org/pdf/2608.23065v1","project":"https://culturalmoment-benchmark.github.io/","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce the Cultural Moment Benchmark (CMB): 306 expert-curated concepts from seven countries in Southeast Asia across five categories.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23065"},"ranking":{"today":{"score":46,"rank":14,"coverage":0.65,"confidence":"Medium"},"30d":{"score":40,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates video cultural reasoning and grounding on 306 expert-curated concepts from Southeast Asia, scoring naming, visual recognition, and temporal localization.","whyItMatters":"Provides a diagnostic harness to pinpoint specific weaknesses in video cultural understanding across different abilities and modalities.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"8795e9f372bfad65ab5448160f285d64b22c791fcf2de2d899c5bf461fe91544"},"motivation":"Cultural understanding in video means more than recognizing what is visible; it requires grasping the symbolic and temporal significance of cultural concepts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark has a public project page, a clear three-stage evaluation protocol, and a large human study, making it accessible and reusable.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce the Cultural Moment Benchmark (CMB)"},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 Main Conference, https://culturalmoment-benchmark","evidence":"Accepted to EMNLP 2026 Main Conference, https://culturalmoment-benchmark.github.io/","evidenceUrl":"http://arxiv.org/abs/2608.23065v1","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-25T16:21:05.134052Z"},"venueAttempts":[{"venueName":"EMNLP 2026 Main Conference, https://culturalmoment-benchmark","reviewStatus":"accepted","decisionRaw":"Accepted to EMNLP 2026 Main Conference, https://culturalmoment-benchmark.github.io/","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"http://arxiv.org/abs/2608.23065v1","observedAt":"2026-08-25T16:21:05.134052Z","rawValue":"Accepted to EMNLP 2026 Main Conference, https://culturalmoment-benchmark.github.io/","level":"author-claim"}]}],"attentionForecast":{"score":66,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses cultural understanding in video, a growing area, and offers a public project page with expert annotations, likely to attract attention from multimodal and cultural AI researchers."},"evaluationMode":"public_reusable","publishers":[{"name":"Cultural Moment Benchmark team","organizationType":"academic-lab","sourceUrl":"https://culturalmoment-benchmark.github.io/","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_cultureforest_c27631e8","familyId":"bmf_2e6f8f580779","name":"CultureForest","oneLine":"CultureForest is a benchmark for cultural norm grounded reasoning, featuring 5,378 examples across 8 domains and 53 countries/regions, with progressive evaluation from multiple-choice to open-ended generation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01879","pdf":"https://arxiv.org/pdf/2606.01879","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01879"},"evidence":{"snippet":"To bridge this gap, we introduce CultureForest, a benchmark for \\textit{Cultural Norm Grounded Reasoning}.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01879"},"ranking":{},"description":"CultureForest is a benchmark for cultural norm grounded reasoning, featuring 5,378 examples across 8 domains and 53 countries/regions, with progressive evaluation from multiple-choice to open-ended generation.","whyItMatters":"The benchmark addresses the evaluation gap between cultural knowledge and its application, offering a verifiable and attributable assessment of reasoning grounded in cultural norms.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2902e9aeb3a587b171ece7f01312000fc53ec4b804513780ce9c1c48b8e84809"},"motivation":"Existing research largely reduces cultural intelligence in LLMs to a knowledge-level problem, overlooking whether models can effectively utilize their acquired knowledge in realistic scenarios.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01879","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_culturetalk-id_d7b2d28e","familyId":"bmf_e0a1bd69f078","name":"CultureTalk-ID","oneLine":"CultureTalk-ID is a dialogue-based benchmark for cultural commonsense in Indonesian and local languages, containing 4,496 culturally grounded dialogues across 11 languages and 13 topics, with three tasks: dialogue-based multiple-choice reasoning, culturally faithful translation, and language steering.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.21016","pdf":"https://arxiv.org/pdf/2607.21016","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.21016"},"evidence":{"snippet":"We introduce CultureTalk-ID, the first dialogue-based benchmark for cultural commonsense in Indonesian and its local languages, comprising 4,496 culturally grounded dialogues across 11 languages and 13 culturally salient topics, curated through a multi-stage human pipeline with native speakers to ensure authenticity.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.21016"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CultureTalk-ID is a dialogue-based benchmark for cultural commonsense in Indonesian and local languages, containing 4,496 culturally grounded dialogues across 11 languages and 13 topics, with three tasks: dialogue-based multiple-choice reasoning, culturally faithful translation, and language steering.","whyItMatters":"Existing cultural benchmarks use isolated prompts; CultureTalk-ID captures cultural nuances in dialogue context, enabling evaluation of models' cultural understanding, transfer, and generation in real conversational settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dd83bfa1dde99e387ddcae17a7fecc17a43d73f72554f26562f0cf44eea78da7"},"motivation":"Culture is lived through conversation, yet existing Indonesian cultural commonsense benchmarks evaluate LLMs on short and isolated prompts, stripping away the dialogic context in which cultural nuances actually surface.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.21016","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"CultureTalk-ID Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.21016","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_culturevidbench_541d6118","familyId":"bmf_3a418ce01d2a","name":"CultureVidBench","oneLine":"CultureVidBench evaluates cultural understanding in text-to-video generation with 1,000 curated prompts covering 12 countries, 6 continents, and 14 cultural aspects. It assesses cultural faithfulness, multimodal cultural rendering, semantic adherence, and perceptual quality using human studies and MLLM-based automatic assessment.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.01942","pdf":"https://arxiv.org/pdf/2608.01942","project":"https://hanxjing.github.io/CultureVidBench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01942"},"evidence":{"snippet":"We introduce CultureVidBench, a comprehensive benchmark for evaluating cultural understanding in T2V generation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01942"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CultureVidBench evaluates cultural understanding in text-to-video generation with 1,000 curated prompts covering 12 countries, 6 continents, and 14 cultural aspects. It assesses cultural faithfulness, multimodal cultural rendering, semantic adherence, and perceptual quality using human studies and MLLM-based automatic assessment.","whyItMatters":"Existing T2V benchmarks focus on perceptual quality and alignment but neglect cultural representation. CultureVidBench addresses this gap by providing a benchmark for evaluating whether generated videos capture culturally specific details, which is crucial for diverse deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dfea5ff7823a68eef3254a2a732074b099e2660dc300cc21629a597a347f64b5"},"motivation":"Text-to-video (T2V) generation models have advanced rapidly, yet their ability to represent diverse cultural contexts remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.01942","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_6e8d6145c32b816c","familyId":"catalog_family_6e8d6145c32b816c","name":"CursorBench","oneLine":"Cursor's current first-party benchmark for ambiguous, multi-file coding-agent tasks from real Cursor sessions.","description":"Cursor's current first-party benchmark for ambiguous, multi-file coding-agent tasks from real Cursor sessions.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://cursor.com/evals","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6e8d6145c32b816c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/cursorbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"cursorBench","url":"https://benchlm.ai/benchmarks/cursorbench","paperUrl":"https://cursor.com/evals","year":"2026","fullName":"CursorBench","format":"Cursor agent-loop evaluation","tasks":"Harder long-horizon agentic coding tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_9b3b3ed1cf7144a8","familyId":"catalog_family_9b3b3ed1cf7144a8","name":"CursorBench v3.2","oneLine":"CursorBench v3.2 evaluates coding agents on interactive software engineering tasks in the Cursor environment.","description":"CursorBench v3.2 evaluates coding agents on interactive software engineering tasks in the Cursor environment.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cursorbench-3.2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9b3b3ed1cf7144a8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cursorbench-3.2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cursorbench-3.2","url":"https://llm-stats.com/benchmarks/cursorbench-3.2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_cv-arena_3558dfa5","familyId":"bmf_cc2f478af1d0","name":"CV-Arena","oneLine":"CV-Arena is a benchmark of 12K high-resolution image instruction pairs across 16 task types, with a human-AI collaborative preference protocol (Active Elo) for evaluating instructional computer vision problem solving.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00931","pdf":"https://arxiv.org/pdf/2606.00931","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.00931"},"evidence":{"snippet":"We introduce CV-Arena, an open benchmark designed to evaluate this capability at professional scales.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00931"},"ranking":{},"description":"CV-Arena is a benchmark of 12K high-resolution image instruction pairs across 16 task types, with a human-AI collaborative preference protocol (Active Elo) for evaluating instructional computer vision problem solving.","whyItMatters":"Addresses the gap in evaluating diverse real-image editing tasks with professional constraints, providing a scalable and traceable evaluation resource for model comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5c20d7f13a506c6bcd4b096345171b0301000890ae3bb6d3c9fa6b1f4d0bbb93"},"motivation":"Instruction-guided image editing is becoming a general interface for visual work, yet existing benchmarks still focus largely on narrow appearance edits and do not fully capture the diversity of real-image tasks in professional workflows.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00931","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_c36c554916db16cd","familyId":"catalog_family_c36c554916db16cd","name":"CVE-Bench zero-day","oneLine":"OpenAI's black-box, no-source variant of CVE-Bench v1 across 40 critical vulnerabilities.","description":"OpenAI's black-box, no-source variant of CVE-Bench v1 across 40 critical vulnerabilities.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://github.com/uiuc-kang-lab/cve-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c36c554916db16cd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/cvebenchzerodayblackbox"}],"catalogSources":[{"catalog":"benchlm","sourceId":"cveBenchZeroDayBlackBox","url":"https://benchlm.ai/benchmarks/cvebenchzerodayblackbox","paperUrl":"https://github.com/uiuc-kang-lab/cve-bench","year":"2026","fullName":"CVE-Bench v1 Zero-Day Black-Box Evaluation","format":"Pass@1 over three rollouts","tasks":"40 critical CVEs","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_cvsbench_ac22ff1a","familyId":"bmf_5b2d8ea2ca58","name":"CVSBench","oneLine":"CVSBench evaluates cross-view spatial reasoning in vision-language models using satellite-street image pairs. It includes tasks for cross-view VQA, grounding, and viewpoint identification, with 3,297 image groups, 9,468 object-level annotations, and 40,679 QA pairs.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22476","pdf":"https://arxiv.org/pdf/2606.22476","project":null,"code":null,"data":"https://huggingface.co/datasets/zlyzlyzly/CVSBench","hfPaper":"https://huggingface.co/papers/2606.22476"},"evidence":{"snippet":"Motivated by this, we introduce CVSBench, a large-scale benchmark for evaluating cross-view spatial reasoning through satellite-street pairs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":968,"hfDatasetLikes":1},"source":{"type":"arxiv","id":"2606.22476"},"ranking":{"90d":{"score":45,"rank":113,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":17,"datasetRankPopulation":66}},"description":"CVSBench evaluates cross-view spatial reasoning in vision-language models using satellite-street image pairs. It includes tasks for cross-view VQA, grounding, and viewpoint identification, with 3,297 image groups, 9,468 object-level annotations, and 40,679 QA pairs.","whyItMatters":"CVSBench addresses the gap in evaluating VLMs' ability to reason about scenes across drastically different viewpoints, which is crucial for applications like navigation and remote sensing. It provides a systematic protocol for measuring object-level and layout consistency under viewpoint changes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"62431a688503ade3ff59c61243b231fb5e9b0ff0aa9e6c6c754b8b2e5d0a2e4a"},"motivation":"Humans can effortlessly reason about scenes across different viewpoints, yet it remains unclear whether Vision-Language Models (VLMs) possess similar cross-view spatial abilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22476","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"earth-insights","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/datasets/zlyzlyzly/CVSBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_0d4bd370b107a31d","familyId":"catalog_family_0d4bd370b107a31d","name":"CVTG-2K","oneLine":"CVTG-2K (Chinese Visual Text Generation 2K) is a benchmark for evaluating text-to-image models on their ability to accurately render text within generated images. It measures Word Accuracy, Normalized Edit Distance (NED), and CLIPScore across 2,000 prompts.","description":"CVTG-2K (Chinese Visual Text Generation 2K) is a benchmark for evaluating text-to-image models on their ability to accurately render text within generated images. It measures Word Accuracy, Normalized Edit Distance (NED), and CLIPScore across 2,000 prompts.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Image-Generation","Language","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cvtg-2k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0d4bd370b107a31d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cvtg-2k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cvtg-2k","url":"https://llm-stats.com/benchmarks/cvtg-2k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image-generation","language","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_cxr-retrieve_95ed172c","familyId":"bmf_7d27f3758eea","name":"CXR-Retrieve","oneLine":"CXR-Retrieve is a benchmark for compositional chest X-ray text-to-image retrieval, with 5,159 MIMIC-CXR-JPG test images and 145 queries over single and conjunction findings with positive and negative assertions.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Information retrieval"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27779","pdf":"https://arxiv.org/pdf/2607.27779","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.27779"},"evidence":{"snippet":"This creates an objective mismatch: a model may retrieve images related to words in the query while failing to satisfy the full clinical constraint, especially for conjunctions and negations such as ``atelectasis and no pneumonia.'' We introduce CXR-Retrieve, a structured benchmark for compositional chest X-ray text-to-image retrieval.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27779"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CXR-Retrieve is a benchmark for compositional chest X-ray text-to-image retrieval, with 5,159 MIMIC-CXR-JPG test images and 145 queries over single and conjunction findings with positive and negative assertions.","whyItMatters":"Addresses the gap between report-to-image matching and complex clinical query satisfaction, crucial for real-world search reliability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"12143ad48a34ee4266801b01b4cb63e6bac96d9a1e0f657123ab864b3f8f11ed"},"motivation":"Large chest radiography archives are difficult to search because most studies are paired only with free-text reports rather than structured clinical annotations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27779","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Search & Retrieval"],"domainScope":"specific"},{"id":"catalog_8bcaca6998006997","familyId":"catalog_family_8bcaca6998006997","name":"Cybench","oneLine":"CyBench is a suite of Capture-the-Flag (CTF) challenges measuring agentic cyber attack capabilities. It evaluates dual-use cybersecurity knowledge and measures the 'unguided success rate', where agents complete tasks end-to-end without guidance on appropriate subtasks.","description":"CyBench is a suite of Capture-the-Flag (CTF) challenges measuring agentic cyber attack capabilities. It evaluates dual-use cybersecurity knowledge and measures the 'unguided success rate', where agents complete tasks end-to-end without guidance on appropriate subtasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Safety","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2408.08926","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8bcaca6998006997"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/cybench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cybench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"cybench","url":"https://benchlm.ai/benchmarks/cybench","paperUrl":"https://arxiv.org/abs/2408.08926","year":"2025","fullName":"Cybench","format":"Cybersecurity agent task completion","tasks":"40 professional CTF tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"cybench","url":"https://llm-stats.com/benchmarks/cybench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","safety","agents","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_b36a9ac751e8e3a6","familyId":"catalog_family_b36a9ac751e8e3a6","name":"CyberBench","oneLine":"Can autonomous agents craft PoC inputs that trigger OSS-Fuzz vulnerabilities—and stop crashing after the fix?","description":"Can autonomous agents craft PoC inputs that trigger OSS-Fuzz vulnerabilities—and stop crashing after the fix?","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/cyber","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b36a9ac751e8e3a6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/cyber"}],"catalogSources":[{"catalog":"benchlm","sourceId":"cyber","url":"https://benchlm.ai/benchmarks/cyber","paperUrl":"https://www.vals.ai/benchmarks/cyber","year":"2026","fullName":"Vals CyberBench","format":"Accuracy score","tasks":"OSS-Fuzz PoC and patch-verification tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_cyberchainbench_89e4f090","familyId":"bmf_073d2b7764c1","name":"CyberChainBench","oneLine":"CyberChainBench evaluates LLM-based agents on smart contract security across vulnerability detection, exploit generation, and patch synthesis, using 541 real exploit incidents with on-chain evaluation on historical forks.","area":"Vision & 3D","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26216","pdf":"https://arxiv.org/pdf/2606.26216","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.26216"},"evidence":{"snippet":"We present CyberChainBench, a benchmark for evaluating LLM-based agents on smart contract security across three complementary tasks: vulnerability detection, exploit generation, and patch synthesis.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26216"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CyberChainBench evaluates LLM-based agents on smart contract security across vulnerability detection, exploit generation, and patch synthesis, using 541 real exploit incidents with on-chain evaluation on historical forks.","whyItMatters":"Smart contract security requires realistic, end-to-end evaluation of agent capabilities; CyberChainBench provides structured ground truth and economic impact metrics to measure practical effectiveness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"372a1ccbf58b9582df8654725de82f7b7ee7945ae56f4e91c642fe524ce71754"},"motivation":"We present CyberChainBench, a benchmark for evaluating LLM-based agents on smart contract security across three complementary tasks: vulnerability detection, exploit generation, and patch synthesis.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.26216","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_d269420d62ea46a3","familyId":"catalog_family_d269420d62ea46a3","name":"CyberGym","oneLine":"CyberGym is a benchmark for evaluating AI agents on cybersecurity tasks, testing their ability to identify vulnerabilities, perform security analysis, and complete security-related challenges in a controlled environment.","description":"CyberGym is a benchmark for evaluating AI agents on cybersecurity tasks, testing their ability to identify vulnerabilities, perform security analysis, and complete security-related challenges in a controlled environment.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Safety","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.cybergym.io/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d269420d62ea46a3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/cybergym"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cybergym"}],"catalogSources":[{"catalog":"benchlm","sourceId":"cyberGym","url":"https://benchlm.ai/benchmarks/cybergym","paperUrl":"https://www.cybergym.io/","year":"2026","fullName":"CyberGym","format":"Vulnerability reproduction and PoC generation","tasks":"1,507 vulnerability analysis instances","successorKey":null},{"catalog":"llm-stats","sourceId":"cybergym","url":"https://llm-stats.com/benchmarks/cybergym","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","safety","agents","code"],"catalogModelCount":13,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_cybergym-e2e_b518f5fa","familyId":"bmf_20d23ee666cc","name":"CyberGym-E2E","oneLine":"CyberGym-E2E evaluates AI agents on end-to-end cybersecurity tasks, covering vulnerability discovery, proof-of-concept generation, and patch generation, using 920 real-world vulnerabilities from 139 open-source projects.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04460","pdf":"https://arxiv.org/pdf/2606.04460","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04460"},"evidence":{"snippet":"To address this gap, we propose CyberGym-E2E, a large-scale and realistic end-to-end cybersecurity benchmark that comprehensively evaluates AI agents' abilities across the full lifecycle of vulnerability discovery, PoC generation, and patch generation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04460"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"CyberGym-E2E evaluates AI agents on end-to-end cybersecurity tasks, covering vulnerability discovery, proof-of-concept generation, and patch generation, using 920 real-world vulnerabilities from 139 open-source projects.","whyItMatters":"Existing cybersecurity benchmarks often lack scale or end-to-end scope, limiting assessment of AI agents' practical utility in vulnerability remediation. CyberGym-E2E aims to address this by providing a large-scale, realistic environment for evaluating agents across the full vulnerability lifecycle.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"77ad3eba7b618827dc512573d320f98e49e16da524e7a34a750085d1d2c611c8"},"motivation":"AI has the potential to transform cybersecurity by enabling systems that can autonomously detect, analyze, and remediate software vulnerabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04460","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_cybermaskqa_2283c85d","familyId":"bmf_5ce23c89d555","name":"CyberMaskQA","oneLine":"The evaluation object is a dataset for privacy-aware cybersecurity question answering, covering key security domains with private entity labels. The main capability evaluated is QA accuracy and masking performance.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24765","pdf":"https://arxiv.org/pdf/2605.24765","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24765"},"evidence":{"snippet":"To address this gap, we introduce CYBERMASKQA, a privacy-aware QA benchmark covering key security domains.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24765"},"ranking":{},"description":"The evaluation object is a dataset for privacy-aware cybersecurity question answering, covering key security domains with private entity labels. The main capability evaluated is QA accuracy and masking performance.","whyItMatters":"The evaluation gap is the lack of context-rich datasets for privacy-preserving QA in cybersecurity, which hinders progress. This benchmark's value is in enabling controlled information disclosure and studying privacy-utility trade-offs for deployable models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"62a9dc9cf905a20e3ecbddde28efeab2638bc9ec43694cb114944b77c4e58a21"},"motivation":"Large language models (LLMs) are increasingly applied to cybersecurity question answering (QA) for critical tasks such as incident response and vulnerability analysis.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24765","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_09526d651b3985e7","familyId":"catalog_family_09526d651b3985e7","name":"CyberSecEval 4","oneLine":"CyberSecEval 4 is an evaluation suite covering cybersecurity-related capabilities and risks of large language models. The insecure-code-generation tracks measure whether a model produces vulnerable code: the Instruct track presents coding requests designed to elicit known insecure patterns, while the Autocomplete track prompts the model with code context leading up to a known insecure pattern, with vulnerabilities detected via static analysis.","description":"CyberSecEval 4 is an evaluation suite covering cybersecurity-related capabilities and risks of large language models. The insecure-code-generation tracks measure whether a model produces vulnerable code: the Instruct track presents coding requests designed to elicit known insecure patterns, while the Autocomplete track prompts the model with code context leading up to a known insecure pattern, with vulnerabilities detected via static analysis.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cyberseceval-4","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_09526d651b3985e7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cyberseceval-4"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cyberseceval-4","url":"https://llm-stats.com/benchmarks/cyberseceval-4","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_d472c7b25fe16c49","familyId":"catalog_family_d472c7b25fe16c49","name":"Cybersecurity CTFs","oneLine":"Cybersecurity Capture the Flag (CTF) benchmark for evaluating LLMs in offensive security challenges. Contains diverse cybersecurity tasks including cryptography, web exploitation, binary analysis, and forensics to assess AI capabilities in cybersecurity problem-solving.","description":"Cybersecurity Capture the Flag (CTF) benchmark for evaluating LLMs in offensive security challenges. Contains diverse cybersecurity tasks including cryptography, web exploitation, binary analysis, and forensics to assess AI capabilities in cybersecurity problem-solving.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/cybersecurity-ctfs","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d472c7b25fe16c49"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/cybersecurity-ctfs"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"cybersecurity-ctfs","url":"https://llm-stats.com/benchmarks/cybersecurity-ctfs","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_6e3a05090425aed5","familyId":"catalog_family_6e3a05090425aed5","name":"CyScenarioBench solved","oneLine":"Number of CyScenarioBench scenarios completed successfully in at least one run.","description":"Number of CyScenarioBench scenarios completed successfully in at least one run.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.irregular.com/research/assessing-gpt-5.6-sol","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6e3a05090425aed5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/cyscenariobenchscenariossolved"}],"catalogSources":[{"catalog":"benchlm","sourceId":"cyScenarioBenchScenariosSolved","url":"https://benchlm.ai/benchmarks/cyscenariobenchscenariossolved","paperUrl":"https://www.irregular.com/research/assessing-gpt-5.6-sol","year":"2026","fullName":"CyScenarioBench Scenarios Ever Solved","format":"Scenarios solved at least once","tasks":"11 cyber scenarios","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_f98db8b9e1e72ce6","familyId":"catalog_family_f98db8b9e1e72ce6","name":"CyScenarioBench success","oneLine":"Average success rate across realistic, long-horizon cybersecurity scenarios.","description":"Average success rate across realistic, long-horizon cybersecurity scenarios.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.irregular.com/research/assessing-gpt-5.6-sol","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f98db8b9e1e72ce6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/cyscenariobenchaveragesuccess"}],"catalogSources":[{"catalog":"benchlm","sourceId":"cyScenarioBenchAverageSuccess","url":"https://benchlm.ai/benchmarks/cyscenariobenchaveragesuccess","paperUrl":"https://www.irregular.com/research/assessing-gpt-5.6-sol","year":"2026","fullName":"CyScenarioBench Average Success Rate","format":"Average success rate","tasks":"11 cyber scenarios","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_d2vbench_a4547106","familyId":"bmf_c0b889ee24cf","name":"D2VBench","oneLine":"D2VBench evaluates LLM value alignment using 10,000 daily dilemma scenarios covering 158 fine-grained value concepts, with a hybrid paradigm of multiple-choice and open-ended questions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19834","pdf":"https://arxiv.org/pdf/2607.19834","project":null,"code":"https://github.com/tjunlp-lab/D2VBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.19834"},"evidence":{"snippet":"To address these issues, we propose D2VBench, a value alignment benchmark comprising 10,000 instances of real daily dilemma scenarios constructed through a multi-stage collaboration between LLMs and humans, grounded in 158 manually annotated fine-grained value concepts.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19834"},"ranking":{"90d":{"score":23,"rank":368,"coverage":0.55,"confidence":"Low"}},"description":"D2VBench evaluates LLM value alignment using 10,000 daily dilemma scenarios covering 158 fine-grained value concepts, with a hybrid paradigm of multiple-choice and open-ended questions.","whyItMatters":"Value alignment benchmarks often lack scenario diversity; this provides a large, fine-grained dataset for assessing alignment across value dimensions in realistic settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fe3c347a785663e72dacd23502fa9078d9753666f45d48c71ad1d40d9c38e963"},"motivation":"With the wide application of large language models (LLMs) in real-world scenarios, the value implication of their outputs is crucial.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19834","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Tianjin University NLP Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/tjunlp-lab/D2VBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_omnijudge-or-omnibias-diagnosing-multimoda_74cb9dd9","familyId":"bmf_664f4b52dc9b","name":"D3-Omni","oneLine":"D3-Omni diagnoses multimodal judge models across T2I, T2V, and TTS tasks with 53 balanced orthogonal binary dimensions and 10,671 samples that isolate specific capability failures.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-26","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.24160","pdf":"https://arxiv.org/pdf/2608.24160","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Motivated by this, we introduce D3-Omni, a balanced and decoupled benchmark for diagnosing fine-grained multimodal understanding, covering 53 orthogonal binary dimensions (17/22/14) and 10,671 samples (3,526/1,998/5,147) across the three tasks.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24160"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"D3-Omni diagnoses multimodal judge models across T2I, T2V, and TTS tasks with 53 balanced orthogonal binary dimensions and 10,671 samples that isolate specific capability failures.","whyItMatters":"It exposes fine-grained blind spots in OmniJudges that aggregate accuracy hides, guiding more reliable automatic evaluation and annotation.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"f733bb137c345d4bddc33b33ef376129d7c3d1f1b92675a8198237abc4961f5b"},"motivation":"Multimodal understanding models that can jointly judge text-to-image (T2I), text-to-video (T2V) and text-to-speech (TTS) generation are increasingly used as \"OmniJudges\" for evaluation and automatic annotation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark provides a stable evaluation protocol with dual-balanced and decoupled design, but no explicit public release statement or artifact link is present, limiting immediate reuse.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce D3-Omni, a balanced and decoupled benchmark for diagnosing fine-grained multimodal understanding"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24160","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":61,"confidence":"Medium","horizon":"7d","reason":"The diagnostic approach to multimodal judge reliability addresses a growing concern, though lack of immediate public artifacts may temper initial engagement."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_dailybench_da2d9ccf","familyId":"bmf_8f2e299ee284","name":"DailyBench","oneLine":"DailyBench is a unified benchmark for evaluating AI-generated image detectors on modern full-image synthesis and object-level manipulation. It comprises FakeBench, with images from recent open-source and commercial generative models, and ManipulationBench, with subtle edits to real images. Detectors are scored by balanced accuracy.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.24016","pdf":"https://arxiv.org/pdf/2607.24016","project":"https://dailybench.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.24016"},"evidence":{"snippet":"To bridge this gap, we introduce DailyBench, a high-quality unified benchmark for evaluating whether AI-generated image detectors can generalize across both modern full-image synthesis and object-level manipulation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24016"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"DailyBench is a unified benchmark for evaluating AI-generated image detectors on modern full-image synthesis and object-level manipulation. It comprises FakeBench, with images from recent open-source and commercial generative models, and ManipulationBench, with subtle edits to real images. Detectors are scored by balanced accuracy.","whyItMatters":"Existing detection benchmarks lag behind current generative models, causing a gap between evaluation and real-world scenarios. DailyBench provides a realistic testbed to assess generalization to modern synthesis and manipulation, offering practical value for developing robust detectors.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"829a2752de5a20de0a54a1781dc1e85fcb0bdf831538d19dd4c1a0eebd9e4c94"},"motivation":"Recent advances in generative models have shifted AI-generated image detection from identifying easily distinguishable, fully synthetic images to identifying highly realistic content generated by both modern generation and manipulation pipelines.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24016","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DailyBench Project","organizationType":"academic-lab","sourceUrl":"https://dailybench.github.io/","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_a10b3e867f137732","familyId":"catalog_family_a10b3e867f137732","name":"DailyOmni","oneLine":"DailyOmni evaluates multimodal models on daily-life video understanding tasks.","description":"DailyOmni evaluates multimodal models on daily-life video understanding tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Video"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/dailyomni","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a10b3e867f137732"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/dailyomni"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"dailyomni","url":"https://llm-stats.com/benchmarks/dailyomni","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","video"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_dailyreport_bfd6c726","familyId":"bmf_5bf51bae2f57","name":"DailyReport","oneLine":"DailyReport is a benchmark with 150 open-ended daily search tasks and 3,546 rubrics, evaluating search agents on information-seeking tasks. Tasks are decomposed into subtasks with cascade rubrics across dimensions, yielding interpretable scores and a user preference score.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.12871","pdf":"https://arxiv.org/pdf/2606.12871","project":null,"code":"https://github.com/AGI-Eval-Official/DailyReport","data":null,"hfPaper":"https://huggingface.co/papers/2606.12871"},"evidence":{"snippet":"To bridge this gap, we introduce DailyReport, an open-ended benchmark to evaluate SA capabilities on daily search tasks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":14,"hfDailySubmittedAt":"2026-06-23T00:00:00.000Z","githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.12871"},"ranking":{"90d":{"score":37,"rank":179,"coverage":0.7,"confidence":"Medium"}},"description":"DailyReport is a benchmark with 150 open-ended daily search tasks and 3,546 rubrics, evaluating search agents on information-seeking tasks. Tasks are decomposed into subtasks with cascade rubrics across dimensions, yielding interpretable scores and a user preference score.","whyItMatters":"Existing search agent benchmarks often use specialized or artificial tasks with coarse scoring, limiting interpretability and real-world relevance. DailyReport provides a more user-centric evaluation protocol for daily search tasks, offering fine-grained, dimension-level scores to inform agent development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b8f41d9137f5c7ed7c1d1328acc7e7a8ba8527f8f34d6acc08bc0cc6a4f0f74c"},"motivation":"Search Agents (SAs) typically leverage large language models (LLMs) to support complex information-seeking tasks by autonomously exploring web sources and synthesizing information into comprehensive responses.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.12871","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_das-bench_4090fd3a","familyId":"bmf_ee75a788da81","name":"DAS-Bench","oneLine":"DAS-Bench provides 30 topics for evaluating academic survey generation, with DAS-Eval scoring 16 criteria across citation quality, taxonomic synthesis, hierarchical discourse, and manuscript reliability.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.18034","pdf":"https://arxiv.org/pdf/2608.18034","project":"https://zhikaixu24.github.io/projects/DAS/","code":"https://github.com/ZhikaiXu24/DAS","data":"https://huggingface.co/datasets/ZhikaiXu24/DAS-2M","hfPaper":null},"evidence":{"snippet":"We further introduce DAS-Bench, a 30-topic benchmark, together with DAS-Eval, which assesses scholarly citation quality, taxonomic synthesis, hierarchical discourse, and manuscript assembly reliability through 16 criteria.","reasonCodes":["exact named benchmark artifact released in abstract","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":null,"githubStars":13,"githubScope":"benchmark_repo","hfDatasetDownloads":373,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2608.18034"},"ranking":{"30d":{"score":41,"rank":48,"coverage":1.0,"confidence":"High","datasetDownloadRank":10,"datasetRankPopulation":30},"90d":{"score":39,"rank":158,"coverage":1.0,"confidence":"High","datasetDownloadRank":25,"datasetRankPopulation":66}},"description":"DAS-Bench provides 30 topics for evaluating academic survey generation, with DAS-Eval scoring 16 criteria across citation quality, taxonomic synthesis, hierarchical discourse, and manuscript reliability.","whyItMatters":"DAS-Bench introduces a standardized, multi-criteria evaluation for publication-oriented survey generation, enabling comparison across systems beyond existing ad hoc assessments.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"e19e61e6954cbfc823d82d53c3f49fad3b03c4b620c27e4495afc91adb16c19e"},"motivation":"Academic surveys play a central role in organizing rapidly expanding scholarly literature, yet their construction requires extensive paper analysis, coherent knowledge organization, fine-grained citation support, and reliable manuscript assembly.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is explicitly named DAS-Bench, released on Hugging Face with evaluation code, and has a stable scoring protocol through DAS-Eval, so it qualifies as a formal public benchmark.","canonicalNameSource":"abstract","canonicalNameEvidence":"We further introduce DAS-Bench, a 30-topic benchmark, together with DAS-Eval, which assesses scholarly citation quality, taxonomic synthesis, hierarchical discourse, and manuscript assembly reliability through 16 criteria."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.18034","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":70,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets automated academic survey generation with released code, data, and leaderboard-style comparisons across state-of-the-art systems, likely drawing interest from research automation communities."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_dataclaweval_e7e41e17","familyId":"bmf_2921fe8e003f","name":"DataClawEval","oneLine":"DataClawEval evaluates autonomous data-engineering agents across 100 end-to-end tasks spanning PySpark, MySQL, HiveSQL, PrestoSQL/Trino, and FlinkSQL, with deterministic rule-based grading in isolated sandboxes.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.28033","pdf":"https://arxiv.org/pdf/2607.28033","project":null,"code":"https://github.com/Dicemy/DataClawEval/tree/master","data":null,"hfPaper":"https://huggingface.co/papers/2607.28033"},"evidence":{"snippet":"To bridge this gap, we introduce DataClawEval, the first comprehensive benchmark designed specifically to evaluate the end-to-end task completion capabilities of autonomous agents in real-world data engineering scenarios.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"hosting_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.28033"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"DataClawEval evaluates autonomous data-engineering agents across 100 end-to-end tasks spanning PySpark, MySQL, HiveSQL, PrestoSQL/Trino, and FlinkSQL, with deterministic rule-based grading in isolated sandboxes.","whyItMatters":"It provides a reproducible, deterministic evaluation for industrial data-engineering workflows, revealing domain-specific strengths and gaps in agent capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"55f381818930f984f7418761c80a05f58e2c49e9f42dbde2060ab18d3f64d1e9"},"motivation":"Large language models (LLMs) and LLM-based agents are increasingly being deployed to automate complex workflows, promising to revolutionize data management and processing.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.28033","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DataClawEval Team","organizationType":"community","sourceUrl":"https://github.com/Dicemy/DataClawEval/tree/master","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_datagovbench_274a71be","familyId":"bmf_07992656535e","name":"DataGovBench","oneLine":"DataGovBench evaluates LLMs on real-world data analysis using government open data. It comprises Table QA (complex decomposable questions with textual or visual answers) and Table Insight (exploratory data analysis producing expert-level findings). Scoring uses unambiguous reference-based metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.06482","pdf":"https://arxiv.org/pdf/2607.06482","project":null,"code":"https://github.com/SoHasegawa/datagovbench","data":null,"hfPaper":"https://huggingface.co/papers/2607.06482"},"evidence":{"snippet":"We introduce DataGovBench, a benchmark derived from governmental open data designed to evaluate LLMs in practical scenarios.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06482"},"ranking":{"90d":{"score":15,"rank":402,"coverage":0.7,"confidence":"Medium"}},"description":"DataGovBench evaluates LLMs on real-world data analysis using government open data. It comprises Table QA (complex decomposable questions with textual or visual answers) and Table Insight (exploratory data analysis producing expert-level findings). Scoring uses unambiguous reference-based metrics.","whyItMatters":"Existing LLM benchmarks miss real-world data complexities like large multi-table datasets and exploratory insight discovery. DataGovBench provides a challenging evaluation to gauge practical readiness of LLMs for data analytics tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"369cde07352dd802105b52d94e627fa762c814a7637f875487176abc53e9914b"},"motivation":"Current benchmarks for evaluating Large Language Models (LLMs) in data analysis often fail to reflect real-world settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06482","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DataGovBench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/SoHasegawa/datagovbench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_datakernelbench_0cd9eeb8","familyId":"bmf_1d6dec07467a","name":"DataKernelBench","oneLine":"Evaluates LLM-generated CUDA or Triton kernels for database query operations on GPUs, measuring execution speedup over torch.compile on TPC-H workloads.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-27","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.25061","pdf":"https://arxiv.org/pdf/2608.25061","project":"https://kerneldf.github.io/datakernelbench","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce DataKernelBench, which translates SQL into validated PyTorch TorchPlan programs and evaluates LLMs that optimize either the core tensor-bounded snippet or the full query in CUDA or Triton through execution-guided repair.","reasonCodes":["coined title prefix ending in Bench or Benchmark","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25061"},"ranking":{"30d":{"score":39,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates LLM-generated CUDA or Triton kernels for database query operations on GPUs, measuring execution speedup over torch.compile on TPC-H workloads.","whyItMatters":"Addresses the lack of benchmarks for database-style operators in LLM kernel generation, enabling comparison of models on real query optimization tasks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"8927d6eddb6caeac73eb74dffbe107a079fe77b3752eea5492a28e7e1afdeac6"},"motivation":"GPUs increasingly accelerate database systems, but query-specific peak performance still often relies on hand-written kernels.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"The paper introduces a named benchmark with specified tasks, evaluation metrics, and a public project page, indicating a stable scoring contract and reuse path.","canonicalNameSource":"paper_title","canonicalNameEvidence":"DataKernelBench: Can LLMs Optimize Database Queries on GPUs?"},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026","evidence":"Accepted at EMNLP 2026. Homepage: https://kerneldf.github.io/datakernelbench","evidenceUrl":"https://arxiv.org/abs/2608.25061","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-27T04:12:10.575570Z"},"venueAttempts":[{"venueName":"EMNLP 2026","reviewStatus":"accepted","decisionRaw":"Accepted at EMNLP 2026. Homepage: https://kerneldf.github.io/datakernelbench","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.25061","observedAt":"2026-08-27T04:12:10.575570Z","rawValue":"Accepted at EMNLP 2026. Homepage: https://kerneldf.github.io/datakernelbench","level":"author-claim"}]}],"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets an emerging intersection of database and GPU optimization with multiple model evaluations and a project page, likely to draw moderate interest from the systems and ML communities."},"evaluationMode":"score_submission","publishers":[{"name":"DataKernelBench Team","organizationType":"academic-lab","sourceUrl":"https://kerneldf.github.io/datakernelbench","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_dataspace_1846702d","familyId":"bmf_2f3146cd78ab","name":"DataSpace","oneLine":"DataSpace evaluates data agents on analytical questions over heterogeneous workspaces containing CSV, JSON, SQLite, Markdown, PDF, and video artifacts. Agents must discover and integrate relevant evidence and return complete tabular results, scored by a deterministic evaluator using header-invariant column alignment and type-aware comparison.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03451","pdf":"https://arxiv.org/pdf/2608.03451","project":null,"code":"https://github.com/HKUSTDial/DataSpace","data":null,"hfPaper":"https://huggingface.co/papers/2608.03451"},"evidence":{"snippet":"We introduce DataSpace, a benchmark in which data agents produce verifiable tabular results from task-local heterogeneous workspaces.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":33,"hfDailySubmittedAt":"2026-08-07T00:00:00.000Z","githubStars":24,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03451"},"ranking":{"30d":{"score":53,"rank":18,"coverage":0.85,"confidence":"High"},"90d":{"score":49,"rank":76,"coverage":0.7,"confidence":"Medium"}},"description":"DataSpace evaluates data agents on analytical questions over heterogeneous workspaces containing CSV, JSON, SQLite, Markdown, PDF, and video artifacts. Agents must discover and integrate relevant evidence and return complete tabular results, scored by a deterministic evaluator using header-invariant column alignment and type-aware comparison.","whyItMatters":"Existing benchmarks isolate structured querying, retrieval, or open-ended analysis. DataSpace unifies evidence discovery, tabular output completeness, and deterministic evaluation, addressing a gap in assessing data agents for real-world analytical tasks. It provides a practical decision value for comparing agent harnesses and multimodal backbones.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"85b53d1fa3d535a851c3dc6cdf57e78343e5aad0a97b9882faeb637accaf314f"},"motivation":"Data agents enable natural-language analytics over organizational workspaces, where relevant evidence may be scattered across databases, structured files, long documents, and multimedia.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03451","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HKUST Dial","organizationType":"academic-lab","sourceUrl":"https://github.com/HKUSTDial/DataSpace","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_dba-bench_29cc8ff8","familyId":"bmf_800372eba78a","name":"DBA-Bench","oneLine":"A production-fidelity benchmark for LLM-based database operations agents, using instrumented PostgreSQL environments with active workloads and outcome-first evaluation across 106 scenarios.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.DB"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.22165","pdf":"https://arxiv.org/pdf/2607.22165","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.22165"},"evidence":{"snippet":"We present DBA-Bench, a benchmark addressing these gaps through production fidelity, outcome-first evaluation, and controlled scenario reproducibility.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.22165"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A production-fidelity benchmark for LLM-based database operations agents, using instrumented PostgreSQL environments with active workloads and outcome-first evaluation across 106 scenarios.","whyItMatters":"Addresses gaps between evaluation and production database operations, offering a reproducible basis for comparing agent safety and efficacy.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3137dc01d4ef08b28fe1e02666d7a60477ec9d9c65946916cc9e7a6ea2f60dc7"},"motivation":"LLM-based database agents show promise, but differing task scopes, testbeds, and metrics hinder comparison.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.22165","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_dblifebench_c72360a2","familyId":"bmf_9c76077fbb9b","name":"DBLifeBench","oneLine":"Evaluates LLMs across five database lifecycle phases using tasks like schema design, SQL implementation, and debugging, with a focus on holistic database management capabilities.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.DB"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03794","pdf":"https://arxiv.org/pdf/2608.03794","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03794"},"evidence":{"snippet":"To bridge this gap, we introduce DBLifeBench, the first benchmark to evaluate LLMs across five critical lifecycle phases: Design, Implementation, Operation, Debugging, and Maintenance.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03794"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLMs across five database lifecycle phases using tasks like schema design, SQL implementation, and debugging, with a focus on holistic database management capabilities.","whyItMatters":"Moves beyond Text-to-SQL to cover the full database lifecycle, revealing trade-offs between specialized and general models and supporting full-stack database intelligence.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"983637e55ee018e991c08d341856fe784b46e695c3ea5eadd94f480a5160d9e2"},"motivation":"Large Language Models (LLMs) are transforming database interaction paradigms, evolving from simple query translators to autonomous database administrators (DBAs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03794","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DBLifeBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.03794","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_evaluating-agentic-code-repair-capabilitie_2495b9d3","familyId":"bmf_54a7d66ae72d","name":"DDBench","oneLine":"DDBench is a code-repair benchmark with 60 historical bugs from 13 open-source distributed systems, evaluated under symptom-only and context-augmented conditions.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.14863","pdf":"https://arxiv.org/pdf/2608.14863","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce DDBench, a code-repair benchmark of 60 historical bugs mined from 13 open-source distributed systems, partitioned into three difficulty tiers.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14863"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"DDBench is a code-repair benchmark with 60 historical bugs from 13 open-source distributed systems, evaluated under symptom-only and context-augmented conditions.","whyItMatters":"It isolates the effect of debugging context on agent success in distributed-system repair, addressing a gap left by single-process benchmarks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"bd71e657f7c9497eb9a36d302c7c393695e50350e6ba4efebb97a59f46a29ce2"},"motivation":"LLM-based coding agents have advanced rapidly on single-process SWE tasks, with frontier models now clustering in the high-70s on SWE-bench Verified.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The abstract formally names and describes the benchmark, and its evaluation protocol is specified with matched conditions.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce DDBench, a code-repair benchmark of 60 historical bugs mined from 13 open-source distributed systems"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14863","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":60,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets a trending agent evaluation niche with a named dataset and comparison conditions, driving moderate attention."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_ddx-trace_5ca2d5dc","familyId":"bmf_282505fdf4da","name":"DDX-TRACE","oneLine":"DDX-TRACE evaluates medical diagnostic trajectories in multimodal neuroradiology over 211 physician-adjudicated cases, where models request imaging studies, update differential diagnoses, and stop with a localized final diagnosis under hidden evidence.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.23629","pdf":"https://arxiv.org/pdf/2605.23629","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.23629"},"evidence":{"snippet":"We introduce DDX-TRACE, a physician-adjudicated benchmark for multimodal neuroradiology that evaluates diagnostic trajectories under hidden evidence over 211 challenging cases.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.23629"},"ranking":{},"description":"DDX-TRACE evaluates medical diagnostic trajectories in multimodal neuroradiology over 211 physician-adjudicated cases, where models request imaging studies, update differential diagnoses, and stop with a localized final diagnosis under hidden evidence.","whyItMatters":"Traditional medical AI benchmarks reward final answers, overlooking workup quality. DDX-TRACE measures evidence-supported diagnostic processes, providing a more realistic and actionable evaluation for clinical decision support.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"3159448f0f991d33571a2865fb65c3ac34606e6b948e262d3bcd2f99d0aff3ca"},"motivation":"Medical diagnosis is not a single prediction from a fully specified vignette.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.23629","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DDX-TRACE Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2605.23629","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_debtbench_7013b5c5","familyId":"bmf_89860188c215","name":"DebtBench","oneLine":"DebtBench evaluates negotiation dialogue systems in debt collection with persona-enriched user profiles to reflect behavioral heterogeneity. It uses a dataset of negotiation scenarios with multiple user personas, scoring financial recovery and interaction experience metrics.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25218","pdf":"https://arxiv.org/pdf/2607.25218","project":null,"code":"https://github.com/YYuHhhh/DebtNegotiation","data":null,"hfPaper":"https://huggingface.co/papers/2607.25218"},"evidence":{"snippet":"To bridge this gap, we propose DebtBench, the first public persona-enriched debt collection benchmark, that highlights behavioral heterogeneity in negotiation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25218"},"ranking":{"90d":{"score":23,"rank":364,"coverage":0.55,"confidence":"Low"}},"description":"DebtBench evaluates negotiation dialogue systems in debt collection with persona-enriched user profiles to reflect behavioral heterogeneity. It uses a dataset of negotiation scenarios with multiple user personas, scoring financial recovery and interaction experience metrics.","whyItMatters":"Existing negotiation benchmarks assume static, rational users, missing real-world behavioral heterogeneity. DebtBench provides a more realistic testbed for financial negotiation, allowing evaluation of models' ability to handle diverse user behaviors and optimize both recovery and user experience.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a048cfa23c639a5575f2662eec4b8c8b210f934092fcaa0797a92ee1d36b2291"},"motivation":"Debt collection is a critical negotiation task in the financial industry, with strong practical relevance and exceptional academic value as a behaviorally rich, high-stakes testbed for human-centered dialogue systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25218","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DebtBench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/YYuHhhh/DebtNegotiation","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_a6a008c7ba1cc43a","familyId":"catalog_family_a6a008c7ba1cc43a","name":"DECK-Bench","oneLine":"DECK-Bench is Moonshot AI's internal evaluation of agents on creating and reasoning over presentation-style knowledge-work artifacts.","description":"DECK-Bench is Moonshot AI's internal evaluation of agents on creating and reasoning over presentation-style knowledge-work artifacts.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Productivity","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a6a008c7ba1cc43a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/deckbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/deck-bench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"deckBench","url":"https://benchlm.ai/benchmarks/deckbench","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"DECK-Bench (Internal)","format":"Internal evaluation score","tasks":"Internal presentation workflows","successorKey":null},{"catalog":"llm-stats","sourceId":"deck-bench","url":"https://llm-stats.com/benchmarks/deck-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","productivity","reasoning","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_decodem_118f6692","familyId":"bmf_7c2b484ff2de","name":"DECODEM","oneLine":"DECODEM provides benchmark datasets for evaluating automated extraction of corporate governance variables from organizational documents, with human-annotated charters and bylaws.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.15879","pdf":"https://arxiv.org/pdf/2607.15879","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.15879"},"evidence":{"snippet":"This paper introduces DECODEM, a set of benchmark datasets for evaluating the automated extraction of corporate governance variables from organizational documents.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.15879"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DECODEM provides benchmark datasets for evaluating automated extraction of corporate governance variables from organizational documents, with human-annotated charters and bylaws.","whyItMatters":"Could support legal research by providing standardized datasets for evaluating extraction methods, but availability is not confirmed.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"80cf46647e89aab5ccb69368bd7f271446d9232a3e15ba051ec1fa73b416dbc8"},"motivation":"Much empirical legal research depends on translating unstructured text into structured variables.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.15879","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_decompbench_f0f37f4f","familyId":"bmf_820b856726b1","name":"DeCompBench","oneLine":"DeCompBench evaluates agent safety against decomposition attacks, where harmful tasks are broken into benign subtasks; it measures refusal rates and objective fulfillment on decomposed variants.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13994","pdf":"https://arxiv.org/pdf/2606.13994","project":null,"code":null,"data":"https://huggingface.co/datasets/decompositionbench/DeCompBench","hfPaper":"https://huggingface.co/papers/2606.13994"},"evidence":{"snippet":"To this end, we introduce DeCompBench, a benchmark designed specifically to evaluate agentic safety under decomposition attacks.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":55,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2606.13994"},"ranking":{"90d":{"score":39,"rank":155,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":61,"datasetRankPopulation":66}},"description":"DeCompBench evaluates agent safety against decomposition attacks, where harmful tasks are broken into benign subtasks; it measures refusal rates and objective fulfillment on decomposed variants.","whyItMatters":"Addresses a security gap not covered by existing agent safety benchmarks, critical for preventing adversarial misuse in deployed agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"13f49fa85a55bff0af3e7ab967aab671bf9160d72b2b6ee6fbd1ecf49e0a55aa"},"motivation":"LLM-based Agents are becoming increasingly capable and widely deployed, creating growing incentives for adversarial misuse in the real-world.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13994","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_deep4ge_710e720a","familyId":"bmf_231023f65b39","name":"Deep4ge","oneLine":"Deep4ge is a dataset of 14,227 training runs from 59 DNN programs with documented faults, providing per-epoch features for fault detection and diagnosis tasks, including binary detection, multi-class diagnosis, and early prediction.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.12868","pdf":"https://arxiv.org/pdf/2607.12868","project":"https://doi.org/10.5281/zenodo.20337241","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.12868"},"evidence":{"snippet":"We present Deep4ge, a controlled benchmark of 14,227 training runs generated from 59 adapted TensorFlow/Keras deep neural network (DNN) programs collected from Stack Overflow.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.12868"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Deep4ge is a dataset of 14,227 training runs from 59 DNN programs with documented faults, providing per-epoch features for fault detection and diagnosis tasks, including binary detection, multi-class diagnosis, and early prediction.","whyItMatters":"This fills the gap of a public dataset for diagnosing DNN implementation faults, supporting reproducible research and tool development for fault detection in deep learning systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b906aff384e402d000e7bfc7b4a236bc02bf36a9c4498a943248d7dbc4a649c2"},"motivation":"Deep learning systems often fail due to subtle implementation faults that alter training behavior.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICSME Data & Tool Track, 2026","evidence":"Accepted at ICSME Data & Tool Track, 2026","evidenceUrl":"https://arxiv.org/abs/2607.12868","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICSME Data & Tool Track, 2026","reviewStatus":"accepted","decisionRaw":"Accepted at ICSME Data & Tool Track, 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.12868","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at ICSME Data & Tool Track, 2026","level":"author-claim"}]}],"publishers":[{"name":"Zenodo","organizationType":"community","sourceUrl":"https://doi.org/10.5281/zenodo.20337241","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_deepbias_70d900bc","familyId":"bmf_dc5ea037d610","name":"DeepBias","oneLine":"DeepBiasBench is a benchmark for in-depth probing of social biases in LVLMs, built using an adaptive framework with dynamic test data generation and iterative rewriting.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CY"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.11228","pdf":"https://arxiv.org/pdf/2607.11228","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.11228"},"evidence":{"snippet":"Furthermore, we build a benchmark named DeepBiasBench using our framework.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.11228"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DeepBiasBench is a benchmark for in-depth probing of social biases in LVLMs, built using an adaptive framework with dynamic test data generation and iterative rewriting.","whyItMatters":"Static bias datasets provide only superficial assessment. DeepBiasBench aims to expose deeper biases through adaptive probing, but its dynamic nature and lack of fixed protocol make it unsuitable for standalone comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a21f1364d73bc7540a039ebc773cc0bd94224880f5fa2a47782329bf551e4f45"},"motivation":"While Large Vision-Language Models (LVLMs) demonstrate remarkable capabilities, they remain highly susceptible to embedded social biases.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.11228","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_deepchart_94c0e719","familyId":"bmf_6a9063c52ce3","name":"DEEPCHART","oneLine":"Faithful chart generation in real-world data-science workflows requires grounding visualizations in scattered evidence, computing chart-ready quantities, and rendering them accurately.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.26757","pdf":"https://arxiv.org/pdf/2608.26757","project":null,"code":"https://github.com/tangdouer1005/DeepChart","data":null,"hfPaper":"https://huggingface.co/papers/2608.26757"},"evidence":{"snippet":"To measure this gap, we introduce DEEPCHART, an expert-annotated benchmark of 1,482 task-conditioned chart-generation instances drawn from real-world scientific papers, financial filings, and ecosystem reports.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26757"},"ranking":{"30d":{"score":23,"rank":107,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":311,"coverage":0.55,"confidence":"Low"}},"motivation":"Faithful chart generation in real-world data-science workflows requires grounding visualizations in scattered evidence, computing chart-ready quantities, and rendering them accurately.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.26757","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_bebcb6daaaabeab9","familyId":"catalog_family_bebcb6daaaabeab9","name":"DeepPlanning","oneLine":"DeepPlanning evaluates LLMs on complex multi-step planning tasks requiring long-horizon reasoning, goal decomposition, and strategic decision-making.","description":"DeepPlanning evaluates LLMs on complex multi-step planning tasks requiring long-horizon reasoning, goal decomposition, and strategic decision-making.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2601.18137","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bebcb6daaaabeab9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/deepplanning"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/deep-planning"}],"catalogSources":[{"catalog":"benchlm","sourceId":"deepPlanning","url":"https://benchlm.ai/benchmarks/deepplanning","paperUrl":"https://arxiv.org/abs/2601.18137","year":"2026","fullName":"DeepPlanning","format":"Long-horizon planning benchmark","tasks":"Travel planning and constrained shopping","successorKey":null},{"catalog":"llm-stats","sourceId":"deep-planning","url":"https://llm-stats.com/benchmarks/deep-planning","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","agents"],"catalogModelCount":9,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_b710395859fc9fe8","familyId":"catalog_family_b710395859fc9fe8","name":"DeepSearchQA","oneLine":"DeepSearchQA is a benchmark for evaluating deep search and question-answering capabilities, testing models' ability to perform multi-hop reasoning and information retrieval across complex knowledge domains.","description":"DeepSearchQA is a benchmark for evaluating deep search and question-answering capabilities, testing models' ability to perform multi-hop reasoning and information retrieval across complex knowledge domains.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Reasoning","Search","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b710395859fc9fe8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/deepsearchqa"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/deepsearchqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"deepSearchQa","url":"https://benchlm.ai/benchmarks/deepsearchqa","paperUrl":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","year":"2026","fullName":"DeepSearchQA","format":"Search / open / find browser-agent evaluation","tasks":"Agentic browsing and list-answer questions","successorKey":null},{"catalog":"llm-stats","sourceId":"deepsearchqa","url":"https://llm-stats.com/benchmarks/deepsearchqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","search","agents"],"catalogModelCount":9,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Search & Retrieval"],"domainScope":"general"},{"id":"bm_deepswe_ebdcd38e","familyId":"bmf_3f3a44e7f6ef","name":"DeepSWE","oneLine":"Evaluates coding agents on 113 original, long-horizon software engineering tasks across 91 open-source repositories in five languages. Tasks are written from scratch, with hand-written verifiers that check requested functionality and accept any correct implementation.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.07946","pdf":"https://arxiv.org/pdf/2607.07946","project":"https://deepswe.datacurve.ai/","code":"https://github.com/datacurve-ai/deep-swe","data":"https://huggingface.co/datasets/datacurve/deep-swe","hfPaper":"https://huggingface.co/papers/2607.07946"},"evidence":{"snippet":"We release the benchmark, its verifiers, and the full record of evaluation trajectories.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-reviewed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":1530,"githubScope":"benchmark_repo","hfDatasetDownloads":1003,"hfDatasetLikes":64},"source":{"type":"arxiv","id":"2607.07946"},"ranking":{"90d":{"score":72,"rank":7,"coverage":1.0,"confidence":"High","datasetDownloadRank":16,"datasetRankPopulation":66}},"description":"Evaluates coding agents on 113 original, long-horizon software engineering tasks across 91 open-source repositories in five languages. Tasks are written from scratch, with hand-written verifiers that check requested functionality and accept any correct implementation.","whyItMatters":"Addresses the gap of benchmarks relying on mined fixes and inherited tests, which can overstate model capability due to pretraining exposure and rigid grading. Provides a reusable evaluation path with verifiers and trajectories for assessing genuine problem-solving ability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e890c3ad2a987f7362993def072c47c833df4d1ceec4c5eb76afd49724188295"},"motivation":"DeepSWE is a benchmark of 113 original, long-horizon software engineering tasks for evaluating coding agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"source-reviewed","reviewedAt":"2026-08-19","sources":["https://datacurve.ai/research","https://deepswe.datacurve.ai/","https://github.com/datacurve-ai/deep-swe","https://arxiv.org/abs/2607.07946"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.07946","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"releaseDates":{"firstPublicAt":"2026-05-18","paperV1At":"2026-07-08"},"publishers":[{"name":"DataCurve","organizationType":"company-research-lab","sourceUrl":"https://deepswe.datacurve.ai/","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"deepSwe","url":"https://benchlm.ai/benchmarks/deepswe","paperUrl":"https://deepswe.datacurve.ai/blog","year":"2026","fullName":"DeepSWE","format":"Pass@1 with confidence interval, cost, time, and token metadata","tasks":"113 software engineering tasks across 91 repositories and 5 languages","successorKey":null},{"catalog":"llm-stats","sourceId":"deepswe","url":"https://llm-stats.com/benchmarks/deepswe","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["external","agents","code"],"catalogModelCount":12,"catalogStarCount":0},{"id":"catalog_eca7687d567a1c3a","familyId":"catalog_family_eca7687d567a1c3a","name":"DeepSWE 1.0","oneLine":"DeepSWE 1.0 pass@1 benchmark as reported by Artificial Analysis using provider-specific harness runs.","description":"DeepSWE 1.0 pass@1 benchmark as reported by Artificial Analysis using provider-specific harness runs.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/deepswe-1.0","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_eca7687d567a1c3a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/deepswe-1.0"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"deepswe-1.0","url":"https://llm-stats.com/benchmarks/deepswe-1.0","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_8498c3e07ce29fb5","familyId":"catalog_family_8498c3e07ce29fb5","name":"DeepSWE 1.1","oneLine":"DeepSWE 1.1 evaluates software engineering agents using the mini-swe-agent harness.","description":"DeepSWE 1.1 evaluates software engineering agents using the mini-swe-agent harness.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/deepswe-1.1","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8498c3e07ce29fb5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/deepswe-1.1"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"deepswe-1.1","url":"https://llm-stats.com/benchmarks/deepswe-1.1","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":27,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_delegateci-bench_0d61203c","familyId":"bmf_3a846b227d7c","name":"DelegateCI-Bench","oneLine":"DelegateCI-Bench evaluates privacy-conscious query rewriting for LLM delegation, with 3,167 samples combining synthetic data across 20 task types, real user queries from WildChat, and a medical challenge set. Systems rewrite queries to suppress non-essential sensitive information.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CR"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04067","pdf":"https://arxiv.org/pdf/2606.04067","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04067"},"evidence":{"snippet":"We introduce DelegateCI-Bench, the first task based Contextual Integrity benchmark for privacy-conscious delegation, comprising 3,167 samples that combine high quality synthetic data spanning 11 tasks and 20 task types, WildChat based real user queries, and a medical challenge set with dense sensitive information.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04067"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DelegateCI-Bench evaluates privacy-conscious query rewriting for LLM delegation, with 3,167 samples combining synthetic data across 20 task types, real user queries from WildChat, and a medical challenge set. Systems rewrite queries to suppress non-essential sensitive information.","whyItMatters":"Addresses a gap in privacy benchmarks by focusing on task-based necessity rather than type-based PII redaction. Measures privacy-utility tradeoff in delegation, supporting safer LLM use.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"96263337428be6b832d99ca1612b8c63f7071289fac80f14187161d10e542d53"},"motivation":"As LLMs become increasingly woven into everyday workflows, user queries sent to cloud hosted LLMs routinely mix task-essential content with task non-essential sensitive disclosures, yet type based PII redaction is context agnostic and may raise two issues: over disclosing untyped sensitive context and over removing answer bearing spans.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04067","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_delistbench_3f7c82bb","familyId":"bmf_24eacb532a62","name":"DelistBench","oneLine":"Evaluates search-enabled LLMs on reconstructing security-level delisting events from public sources, using a 1,200-record benchmark with joint accuracy metrics.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.22770v1","pdf":"https://arxiv.org/pdf/2608.22770v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce Search-to-Record, a database-assurance task in which search-enabled large language models reconstruct institution-defined event records from public sources for a known security universe and historical cutoff, and DelistBench, a 1,200-record benchmark for security-level delisting announcements.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22770"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates search-enabled LLMs on reconstructing security-level delisting events from public sources, using a 1,200-record benchmark with joint accuracy metrics.","whyItMatters":"Provides a standardized way to assess database assurance for corporate-event completeness, with practical guidance on model selection and triage.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"351f2c39e6980a792ce99a4e5efa8306faba5149b83a4b4a752612eae00ed6f8"},"motivation":"Financial institutions need an independent way to detect missing, stale, and misclassified corporate-event records in vendor databases.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark defines a clear task, provides a fixed dataset, and evaluates multiple models under controlled conditions, making it reusable for further comparisons.","canonicalNameSource":"abstract","canonicalNameEvidence":"and DelistBench, a 1,200-record benchmark for security-level delisting announcements."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22770v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:42:57.448509Z"},"attentionForecast":{"score":54,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses financial data quality and compares web-enabled LLMs, a topic of interest to both finance and NLP communities, but with narrow domain appeal."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_deltaml-bench_b6518e0f","familyId":"bmf_dcd3591b7e50","name":"DeltaML-Bench","oneLine":"DeltaML-Bench contains 48 tasks from research papers requiring agents to improve published baselines within open-source repositories, evaluated under compute constraints with fixed grading and integrity checks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-21","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.19653","pdf":"https://arxiv.org/pdf/2608.19653","project":null,"code":"https://github.com/AlgorithmicResearchGroup/deltaml-bench-public","data":null,"hfPaper":null},"evidence":{"snippet":"We introduce DeltaML-Bench, a benchmark comprising 48 tasks sourced from research papers that require agents to improve published baselines within imperfect, open-source repositories.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.19653"},"ranking":{"30d":{"score":23,"rank":145,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":349,"coverage":0.55,"confidence":"Low"}},"description":"DeltaML-Bench contains 48 tasks from research papers requiring agents to improve published baselines within open-source repositories, evaluated under compute constraints with fixed grading and integrity checks.","whyItMatters":"It captures realistic conditions for autonomous ML experimentation, including heterogeneous codebases and specification gaming risks, providing a more faithful testbed for agent evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-25T16:23:14.196185Z","inputHash":"1cf16b69563516963cfef761bd4cd5c6b1a32eae53bed6a0f2bda2af3ddb8a12"},"motivation":"Autonomous agents for machine learning experimentation must navigate heterogeneous repositories, repair training pipelines, and evaluate candidate improvements under realistic compute constraints.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T15:38:21.596820Z","model":"deepseek-v4-pro","decisionReason":"Unclear official benchmark name: paper uses DeltaML-Bench but repository is named DeltaMLBench; formal release needs confirmation from paper or README.","canonicalNameSource":"paper_title","canonicalNameEvidence":"DeltaML-Bench: Evaluating Machine Learning Agents on Real-World Research Repositories"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.19653","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-21T04:30:09.569171Z"},"evaluationMode":"score_submission","publishers":[{"name":"Algorithmic Research Group","organizationType":"benchmark-organization","sourceUrl":"https://github.com/AlgorithmicResearchGroup/deltaml-bench-public","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_delverl_572c8749","familyId":"bmf_0c3fd3592789","name":"DelveRL","oneLine":"Evaluates local game-playing agents on a deterministic, procedurally generated, partially observed roguelike where agents must secure a key and reach the exit on each floor, scored by floor depth under a 1,500-turn cap.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/SnyderConsulting/DelveRL","pdf":null,"project":"https://godotengine.org/download/","code":"https://github.com/SnyderConsulting/DelveRL","data":null,"hfPaper":null},"evidence":{"snippet":"DelveRL A human-playable turn-based roguelike and open benchmark for local game-playing agents.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:snyderconsulting/delverl"},"ranking":{"30d":{"score":28,"rank":89,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":259,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates local game-playing agents on a deterministic, procedurally generated, partially observed roguelike where agents must secure a key and reach the exit on each floor, scored by floor depth under a 1,500-turn cap.","whyItMatters":"Provides a reproducible, renderer-independent environment for testing long-horizon planning and exploration under partial observability, with a released baseline and deterministic audits to anchor comparisons.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"b5b87cb24da39ff50864875c14fd8c0be69942c0e3517345583252786682c6da"},"motivation":"DelveRL A human-playable turn-based roguelike and open benchmark for local game-playing agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/SnyderConsulting/DelveRL","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":40,"confidence":"Low","horizon":"7d","reason":"The niche roguelike agent benchmark targets a narrow local-agent audience and lacks a linked paper or established community, limiting expected early attention."},"evaluationMode":"score_submission","publishers":[{"name":"Snyder Consulting","organizationType":"company-research-lab","sourceUrl":"https://github.com/SnyderConsulting/DelveRL","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_deploybench_34fae562","familyId":"bmf_c58205378bb9","name":"DeployBench","oneLine":"DeployBench evaluates LLM agents on 51 research-artifact deployment tasks across AI/ML, systems, and scientific computing. Tasks require setting up environments with multi-language toolchains and system-level dependencies, verified by hidden pipelines that execute the paper's experiments and check outputs.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05238","pdf":"https://arxiv.org/pdf/2606.05238","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05238"},"evidence":{"snippet":"We introduce DeployBench, a multi-domain benchmark of 51 research-artifact deployment tasks spanning AI/ML, computer systems, and scientific computing, covering all these dimensions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05238"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DeployBench evaluates LLM agents on 51 research-artifact deployment tasks across AI/ML, systems, and scientific computing. Tasks require setting up environments with multi-language toolchains and system-level dependencies, verified by hidden pipelines that execute the paper's experiments and check outputs.","whyItMatters":"Current benchmarks overlook the complexity of deploying research artifacts, which is a bottleneck for reproducibility. DeployBench provides a standardized testbed with hidden verification, enabling comparison of agents on a realistic deployment task and highlighting failures in completion-judgment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"83ba1a8ef38971c20851ed0f22a306032c3a9d4e36c920641ef874620efd398c"},"motivation":"LLM agents have made rapid progress on software engineering and ML research tasks, but these advances often assume access to a working runnable environment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05238","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_00f9bfbf6e386381","familyId":"catalog_family_00f9bfbf6e386381","name":"DermMCQA","oneLine":"Dermatology multiple choice question assessment benchmark for evaluating medical knowledge and diagnostic reasoning in dermatological conditions and treatments.","description":"Dermatology multiple choice question assessment benchmark for evaluating medical knowledge and diagnostic reasoning in dermatological conditions and treatments.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/dermmcqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_00f9bfbf6e386381"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/dermmcqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"dermmcqa","url":"https://llm-stats.com/benchmarks/dermmcqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["healthcare"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_09a0b59ec4742a3e","familyId":"catalog_family_09a0b59ec4742a3e","name":"Design Arena Agentic Web Dev","oneLine":"A display-only Elo rating from blinded comparisons of multi-file web applications built by coding agents.","description":"A display-only Elo rating from blinded comparisons of multi-file web applications built by coding agents.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://intelligence.ai/leaderboard/webapps","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_09a0b59ec4742a3e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/designarenaagenticwebdev"}],"catalogSources":[{"catalog":"benchlm","sourceId":"designArenaAgenticWebDev","url":"https://benchlm.ai/benchmarks/designarenaagenticwebdev","paperUrl":"https://intelligence.ai/leaderboard/webapps","year":"2026","fullName":"Design Arena Agentic Web Dev Elo","format":"Elo from blinded human preferences","tasks":"Multi-file web application development","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0d1002578c1f2093","familyId":"catalog_family_0d1002578c1f2093","name":"Design Arena Website","oneLine":"A display-only Design Arena website-generation Elo score surfaced on OpenRouter model benchmark pages.","description":"A display-only Design Arena website-generation Elo score surfaced on OpenRouter model benchmark pages.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openrouter.ai/x-ai/grok-4.3/benchmarks","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0d1002578c1f2093"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/designarenawebsite"}],"catalogSources":[{"catalog":"benchlm","sourceId":"designArenaWebsite","url":"https://benchlm.ai/benchmarks/designarenawebsite","paperUrl":"https://openrouter.ai/x-ai/grok-4.3/benchmarks","year":"2026","fullName":"Design Arena Website Elo","format":"Elo","tasks":"Website generation comparisons","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_6acfcea734e55ac9","familyId":"catalog_family_6acfcea734e55ac9","name":"Design2Code","oneLine":"Design2Code evaluates the ability to generate code (HTML/CSS/JS) from visual designs.","description":"Design2Code evaluates the ability to generate code (HTML/CSS/JS) from visual designs.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Code","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6acfcea734e55ac9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/design2code"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/design2code"}],"catalogSources":[{"catalog":"benchlm","sourceId":"design2Code","url":"https://benchlm.ai/benchmarks/design2code","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"Design2Code","format":"Visual input to frontend implementation","tasks":"Design-to-code tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"design2code","url":"https://llm-stats.com/benchmarks/design2code","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","code","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_deskcraft_5fe792c2","familyId":"bmf_eaf217e35324","name":"DeskCraft","oneLine":"DeskCraft evaluates desktop GUI agents on long-horizon professional workflows and human-in-the-loop collaboration across 538 executable tasks in live Ubuntu desktop environments. Covers design, video, audio, and 3D creation software, with execution-based verification.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03103","pdf":"https://arxiv.org/pdf/2606.03103","project":null,"code":"https://github.com/mrwwk/DeskCraft","data":null,"hfPaper":"https://huggingface.co/papers/2606.03103"},"evidence":{"snippet":"To address this issue, we introduce DeskCraft, a desktop GUI benchmark targeting long horizon creative and engineering workflows and proactive human-agent collaboration.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":null,"githubStars":91,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03103"},"ranking":{"90d":{"score":53,"rank":48,"coverage":0.7,"confidence":"Medium"}},"description":"DeskCraft evaluates desktop GUI agents on long-horizon professional workflows and human-in-the-loop collaboration across 538 executable tasks in live Ubuntu desktop environments. Covers design, video, audio, and 3D creation software, with execution-based verification.","whyItMatters":"Existing desktop benchmarks simplify tasks and lack human-agent interaction. DeskCraft measures agent performance on realistic workflows and proactive collaboration, identifying gaps in long-horizon delivery and clarification.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9197fd4bd53ae02283a924b4d2474a32ca619a045171b712ab56fadefa7887db"},"motivation":"Real-world professional desktop workflows in specialized creative and engineering software unfold over long horizons and often require human-in-the-loop coordination, where agents proactively seek necessary information and users provide additional instructions, clarifications, feedback, or corrections as the task progresses.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03103","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DeskCraft team","organizationType":"community","sourceUrl":"https://github.com/mrwwk/DeskCraft","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_desktop-delta-bench_ac655444","familyId":"bmf_40d33a605fa0","name":"Desktop-Delta Bench","oneLine":"Desktop-Delta Bench (DDB) evaluates computer-use models on step-level GUI transition understanding through two tasks: temporal ordering of 3-frame observations and before-after pair classification with five action types, covering 2,013 human-verified instances across ~15 applications.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.26041","pdf":"https://arxiv.org/pdf/2607.26041","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.26041"},"evidence":{"snippet":"We introduce Desktop-Delta Bench (DDB), an offline step-level benchmark with 2,013 human-verified instances from novel, multi-app Linux trajectories across ~15 applications and 50 task domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26041"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Desktop-Delta Bench (DDB) evaluates computer-use models on step-level GUI transition understanding through two tasks: temporal ordering of 3-frame observations and before-after pair classification with five action types, covering 2,013 human-verified instances across ~15 applications.","whyItMatters":"Existing benchmarks focus on end-task success or single-frame grounding, missing the ability to reconstruct causal transitions. DDB provides a diagnostic layer for state verification, source tracking, and context-aware control, enabling targeted improvements in desktop CUA reliability and recovery.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"48ef5f046f9fa5edccd3ad09ee29b94020f44722690242ad66d15d01f536730a"},"motivation":"Computer-use agents (CUAs) increasingly act through desktop GUIs to complete long-horizon tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26041","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_devicesworld_1a9e0be1","familyId":"bmf_c77bf3b044f4","name":"DevicesWorld","oneLine":"DevicesWorld is an executable benchmark for cross-device agent evaluation, comprising 6,140 tasks across Android, Linux, and SmartHome environments, with rule-based verifiers for automatic scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.13465","pdf":"https://arxiv.org/pdf/2607.13465","project":null,"code":"https://github.com/AgenticOrgLab/DevicesWorld","data":null,"hfPaper":"https://huggingface.co/papers/2607.13465"},"evidence":{"snippet":"We introduce DevicesWorld, a large-scale executable benchmark for cross-device collaborative operation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.13465"},"ranking":{"90d":{"score":33,"rank":217,"coverage":0.55,"confidence":"Low"}},"description":"DevicesWorld is an executable benchmark for cross-device agent evaluation, comprising 6,140 tasks across Android, Linux, and SmartHome environments, with rule-based verifiers for automatic scoring.","whyItMatters":"Existing benchmarks focus on single-device environments, leaving a gap in evaluating agents that must coordinate across heterogeneous devices. DevicesWorld aims to address this by providing tasks with cross-device dependencies and automated verification, which could inform development of more capable multi-device agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bb7b831b85e4996ca18d28a1b11d156e22ee3aeca761fac80c556b3f321ae5b0"},"motivation":"LLM-based agents have rapidly improved at operating individual digital environments such as mobile applications, desktop systems, and smart homes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.13465","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_dexverse_baeb3cfa","familyId":"bmf_e243a022bf00","name":"DexVerse","oneLine":"Evaluates dexterous manipulation across 100 tasks, multiple embodiments, and visual variations, with 3,180 demonstrations and a VR teleoperation interface.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.08751","pdf":"https://arxiv.org/pdf/2607.08751","project":"https://ycyao216.github.io/DexVerse.site","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.08751"},"evidence":{"snippet":"We present DexVerse, a large-scale and modular benchmark for dexterous manipulation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.08751"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates dexterous manipulation across 100 tasks, multiple embodiments, and visual variations, with 3,180 demonstrations and a VR teleoperation interface.","whyItMatters":"Provides a comprehensive testbed for studying cross-task and cross-embodiment generalization in dexterous manipulation, addressing gaps in existing benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f915a8df24c6c7054ed0e481048ad2cb383cd9cdc5fd6a895172af61f1d8ada1"},"motivation":"Building general-purpose dexterous manipulation policies requires benchmarks that go beyond isolated tasks to systematically evaluate policies across diverse interaction modes, sensory conditions, and robot embodiments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.08751","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_dgx-spark-llm-lab_ca498b49","familyId":"bmf_cee898b42505","name":"dgx-spark-llm-lab","oneLine":"This repository contains tools for benchmarking LLM configurations (model, quantization, serving flags, thinking mode) on local hardware, with hidden executable tests and leaderboard-like reports.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/luongnv89/dgx-spark-llm-lab","pdf":null,"project":"https://img.shields.io/badge/license-MIT-blue","code":"https://github.com/luongnv89/dgx-spark-llm-lab","data":null,"hfPaper":null},"evidence":{"snippet":"Benchmark model + quant + serving flags + thinking mode on your own hardware, then install the config that won.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:luongnv89/dgx-spark-llm-lab"},"ranking":{"30d":{"score":23,"rank":144,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":348,"coverage":0.55,"confidence":"Low"}},"description":"This repository contains tools for benchmarking LLM configurations (model, quantization, serving flags, thinking mode) on local hardware, with hidden executable tests and leaderboard-like reports.","whyItMatters":"It provides a method for finding optimal LLM setups on specific hardware, but it is a tool/leaderboard rather than a formal benchmark with a fixed evaluation object.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-25T16:23:14.196185Z","inputHash":"6c122783c9207af43e22d9cf8a8207fd946bc2927cb2ff6bf3530c77877af2d1"},"motivation":"dgx-spark-llm-lab Not a leaderboard — a way to pick your daily driver.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T15:38:21.596820Z","model":"deepseek-v4-pro","decisionReason":"Lacks a formally declared benchmark name and acts as an internal evaluation tool rather than a public benchmark for model comparison."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/luongnv89/dgx-spark-llm-lab","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_diagchain_064ef8b0","familyId":"bmf_818fcbc7cc94","name":"DiagChain","oneLine":"DiagChain evaluates LLM agents on evidence-grounded attack chain reconstruction from heterogeneous telemetry. It provides 69 scenarios and five metrics for stage-wise assessment of reasoning steps.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03591","pdf":"https://arxiv.org/pdf/2608.03591","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03591"},"evidence":{"snippet":"We present DiagChain, a diagnostic benchmark for evidence-grounded attack chain reconstruction that enables stage-wise evaluation of LLM agents.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03591"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DiagChain evaluates LLM agents on evidence-grounded attack chain reconstruction from heterogeneous telemetry. It provides 69 scenarios and five metrics for stage-wise assessment of reasoning steps.","whyItMatters":"Existing benchmarks often report end-to-end accuracy, hiding where errors arise. DiagChain enables systematic diagnosis of intermediate stages, offering actionable insights for improving cybersecurity agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cee8f4728507ea98f957ed73f00e42d1816a5d20d4530da2c45f8ed06d2adf58"},"motivation":"Large Language Model (LLM) agents offer a promising approach to attack chain reconstruction by retrieving and interpreting heterogeneous telemetry to infer ordered attacker actions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03591","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_diagram-mmu_a41bffa4","familyId":"bmf_39ca0919aed5","name":"Diagram-MMU","oneLine":"Diagram-MMU evaluates multimodal language models on parsing scientific diagrams into LaTeX TikZ code, editing diagram code, and answering questions about diagrams, using 3.7k diagrams and 18.3k questions across six domains.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12262","pdf":"https://arxiv.org/pdf/2608.12262","project":"https://vi-ocean.github.io/projects/diagram-mmu","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12262"},"evidence":{"snippet":"In this paper, we build a benchmark, Diagram-MMU, a multi-modal benchmark designed to assess MLLMs' ability for scientific diagram parsing and understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12262"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Diagram-MMU evaluates multimodal language models on parsing scientific diagrams into LaTeX TikZ code, editing diagram code, and answering questions about diagrams, using 3.7k diagrams and 18.3k questions across six domains.","whyItMatters":"The evaluation provides insight into model capabilities on diagram-to-code generation tasks, a practical need in scientific authoring tools, and identifies gaps between reasoning and code generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1adae049c186596aba9d2f70186ac67ca1a4334d31facfbf4a63cddbd6526ca2"},"motivation":"Multimodal Large Language Models (MLLMs) have been growing the capability for scientific writing and collaboration.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12262","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_dicobench_558e401f","familyId":"bmf_1608a2621237","name":"DiCoBench","oneLine":"DiCoBench evaluates multimodal LLMs on multi-image fine-grained perception using high-resolution (up to 2K) image pairs. It includes 765 multiple-choice questions across two tracks: differential and commonality visual cues, covering 8 perception tasks, with exact-match scoring.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26602","pdf":"https://arxiv.org/pdf/2606.26602","project":null,"code":"https://github.com/PKU-ICST-MIPL/DICO_Bench_ECCV2026","data":null,"hfPaper":"https://huggingface.co/papers/2606.26602"},"evidence":{"snippet":"To bridge this gap, we introduce DiCoBench, a comprehensive, multi-image high-resolution benchmark designed for cross-image fine-grained perception.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26602"},"ranking":{"90d":{"score":17,"rank":394,"coverage":0.7,"confidence":"Medium"}},"description":"DiCoBench evaluates multimodal LLMs on multi-image fine-grained perception using high-resolution (up to 2K) image pairs. It includes 765 multiple-choice questions across two tracks: differential and commonality visual cues, covering 8 perception tasks, with exact-match scoring.","whyItMatters":"Existing benchmarks rely on explicit textual cues or low resolution, whereas DiCoBench targets autonomous discovery of subtle visual cues in high-resolution pairs. It provides a challenging testbed with a large human-model performance gap, aiding progress in complex multi-image understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"057c225e8c3eaf519674e7e2633fffa5561c01c21c064990aba6ff97ba03bc8e"},"motivation":"Recent advancements in Multimodal Large Language Models (MLLMs) have demonstrated impressive fine-grained perception capabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026","evidence":"Accepted by ECCV 2026. Project page with code: https://github.com/PKU-ICST-MIPL/DICO_Bench_ECCV2026","evidenceUrl":"https://arxiv.org/abs/2606.26602","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026","reviewStatus":"accepted","decisionRaw":"Accepted by ECCV 2026. Project page with code: https://github.com/PKU-ICST-MIPL/DICO_Bench_ECCV2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.26602","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by ECCV 2026. Project page with code: https://github.com/PKU-ICST-MIPL/DICO_Bench_ECCV2026","level":"author-claim"}]}],"publishers":[{"name":"PKU-ICST-MIPL","organizationType":"academic-lab","sourceUrl":"https://github.com/PKU-ICST-MIPL/DICO_Bench_ECCV2026","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_dig-bench_37529e7b","familyId":"bmf_78f64581e492","name":"DiG-bench","oneLine":"Evaluates AI agents on discovery of unknown transformation rules across 70 independent games with seven difficulty tiers, requiring experimentation and rule generalization.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.12593","pdf":"https://arxiv.org/pdf/2608.12593","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12593"},"evidence":{"snippet":"To address this gap, we release a new benchmark: DiG-bench (Discovery in Games).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12593"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates AI agents on discovery of unknown transformation rules across 70 independent games with seven difficulty tiers, requiring experimentation and rule generalization.","whyItMatters":"Fills the gap in benchmarks for scientific discovery in controlled environments, testing the capacity to formulate novel generalizations through interaction.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"8b088888b08ff4e9e027e58c7b770b09d818da4c50254aac13ffcf3e287c7c8f"},"motivation":"Discovery---formulating novel generalizations---is a central part of the scientific process.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12593","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DiG-bench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.12593","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_directorbench_ecf5066a","familyId":"bmf_02198acaf891","name":"DirectorBench","oneLine":"DirectorBench evaluates long-form video generation across 5 dimensions (script, visual, audio, cross-modal, stability) using 80 metadata entries, 7 user profiles, and 40 checkpoint criteria.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.30090","pdf":"https://arxiv.org/pdf/2605.30090","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30090"},"evidence":{"snippet":"We introduce DirectorBench, a personalized multi-agent diagnostic benchmark for long-form video generation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30090"},"ranking":{},"description":"DirectorBench evaluates long-form video generation across 5 dimensions (script, visual, audio, cross-modal, stability) using 80 metadata entries, 7 user profiles, and 40 checkpoint criteria.","whyItMatters":"Current video benchmarks focus on short clips and aggregate scores, missing workflow failures and user preferences. DirectorBench provides diagnostic, profile-aware evaluation to guide improvements in long-form video generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9fc5422221b76b67fcfeb6a02b2898ca0e9f60e6ee23fa1997e0487b04bad29d"},"motivation":"Long-form video generation is rapidly moving from short, single-scene synthesis toward minute-long, multi-shot creation with narrative structure, cinematic control, audio, and cross-modal synchronization.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30090","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_disasterbench_325517be","familyId":"bmf_70b465e708e7","name":"DisasterBench","oneLine":"Evaluates multimodal reasoning for UAV-based disaster response across 14 scene types and 9 tasks spanning pre-, during-, and post-disaster stages.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06217","pdf":"https://arxiv.org/pdf/2606.06217","project":null,"code":"https://github.com/TanmouTT/DisasterBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.06217"},"evidence":{"snippet":"We introduce DisasterBench, a multi-stage multimodal reasoning benchmark for UAV-Based disaster response in complex environments.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06217"},"ranking":{"90d":{"score":37,"rank":175,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates multimodal reasoning for UAV-based disaster response across 14 scene types and 9 tasks spanning pre-, during-, and post-disaster stages.","whyItMatters":"Addresses the lack of benchmarks covering multi-stage disaster reasoning with causal attribution, prediction, and decision-making under low-altitude UAV views and on-site compute constraints, offering a means to compare models on practical emergency-response tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6385868180809c00bd17edb1b048ea478c407e6d133884a6037901cfe7a9cddc"},"motivation":"When a disaster unfolds, responders must answer not only what is happening, but also why it is happening, what will happen next, and what to do now, often from noisy low-altitude UAV views and under tight on-site compute constraints.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.06217","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TanmouTT/DisasterBench","organizationType":"community","sourceUrl":"https://github.com/TanmouTT/DisasterBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_disasterbench_d2342204","familyId":"bmf_70b465e708e7","name":"DisasterBench","oneLine":"DisasterBench is a benchmark for evaluating structured multi-agent planning over disaster-response tools. It includes 233 expert-verified tasks, 26 agents, 81 typed compatibility edges, and 5 planning paradigms. It uses First-Point-of-Failure (FPoF) for step-level failure attribution.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2605.27957","pdf":"https://arxiv.org/pdf/2605.27957","project":null,"code":"https://github.com/TamuChen18/DisasterBench_Open","data":null,"hfPaper":"https://huggingface.co/papers/2605.27957"},"evidence":{"snippet":"We introduce DisasterBench, a benchmark for evaluating structured multi-agent planning over semantically similar but operationally distinct disaster-response tools.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27957"},"ranking":{},"description":"DisasterBench is a benchmark for evaluating structured multi-agent planning over disaster-response tools. It includes 233 expert-verified tasks, 26 agents, 81 typed compatibility edges, and 5 planning paradigms. It uses First-Point-of-Failure (FPoF) for step-level failure attribution.","whyItMatters":"Disaster response requires orchestrating heterogeneous AI tools into executable workflows. DisasterBench tests grounding under typed interface constraints, highlighting gaps between semantic reasoning and execution consistency, and providing diagnostics for failure attribution.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f07ac4e1004cd46c3a46982e0796b8416bcf07bc68373ac843ced15fc3ba003d"},"motivation":"Disasters cause severe societal impacts, demanding rapid coordination of heterogeneous AI tools, from satellite analysis to flood prediction and damage assessment, into coherent multi-step workflows.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27957","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DisasterBench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/TamuChen18/DisasterBench_Open","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_discobench_0216743e","familyId":"bmf_d21d9bcb5de7","name":"DiscoBench","oneLine":"DiscoBench evaluates search agents on clarification-aware deep search, covering 211 samples and 463 ambiguity instances across 11 domains, with four ambiguity types, measuring task utility, ambiguity detection, interaction strategy, and cost efficiency.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.27669","pdf":"https://arxiv.org/pdf/2606.27669","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.27669"},"evidence":{"snippet":"To address this gap, we introduce DiscoBench, a benchmark for clarification-aware deep search, designed to evaluate whether search agents can proactively identify ambiguity, ask effective clarification questions, and recover correct reasoning paths through user interaction.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":16,"hfDailySubmittedAt":"2026-07-03T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.27669"},"ranking":{"90d":{"score":50,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"DiscoBench evaluates search agents on clarification-aware deep search, covering 211 samples and 463 ambiguity instances across 11 domains, with four ambiguity types, measuring task utility, ambiguity detection, interaction strategy, and cost efficiency.","whyItMatters":"This benchmark addresses the gap in evaluating search agents' ability to handle ambiguous and underspecified queries, which is common in real-world search. It assesses proactive clarification and interaction efficiency, offering practical value for improving agent decision-making in complex information-seeking tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f92a3e3484e5f9ff9bdafbe536e98d73eeef3d8c92884d2c801a9fa46840e3fd"},"motivation":"Search agents powered by large language models (LLMs) are increasingly used to solve complex information-seeking tasks, requiring multi-step retrieval and reasoning to fulfill user goals.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.27669","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_discoverphysics_9ae7e548","familyId":"bmf_48d63b5c443b","name":"DiscoverPhysics","oneLine":"DiscoverPhysics is an interactive benchmark for LLM agents to discover laws of motion in simulated worlds with altered physics, scoring trajectory MSE and explanation quality.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26087","pdf":"https://arxiv.org/pdf/2605.26087","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.26087"},"evidence":{"snippet":"We introduce DiscoverPhysics, an interactive benchmark that asks a LLM agent to discover the laws of motion of a simulated world whose physics deliberately deviates from our own.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26087"},"ranking":{},"description":"DiscoverPhysics is an interactive benchmark for LLM agents to discover laws of motion in simulated worlds with altered physics, scoring trajectory MSE and explanation quality.","whyItMatters":"It probes long-horizon reasoning and hypothesis refinement, distinguishing recall from genuine scientific reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"17d1615f8ab47ce652f2f499d346be2d6771788446ac84148b256c1addd7b314"},"motivation":"Frontier LLMs now perform strongly across a wide range of physics evaluations, but it is hard to disentangle genuine reasoning from recall of established science.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26087","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_distract-bench_3873c48a","familyId":"bmf_74ca191db947","name":"Distract-Bench","oneLine":"Distract-Bench evaluates robustness of vision-language models to semantic visual distractions, which are meaningful but task-irrelevant cues that preserve the ground-truth answer.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Robustness"],"topics":["Multimodal","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08894","pdf":"https://arxiv.org/pdf/2606.08894","project":null,"code":"https://github.com/Yizheng-Sun/Distract-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.08894"},"evidence":{"snippet":"To address this gap, we introduce \\textbf{Distract-Bench}, a benchmark for evaluating VLM robustness to \\textbf{semantic visual distractions}, defined as meaningful but task-irrelevant visual cues added to inputs while preserving the ground-truth answer.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":10,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08894"},"ranking":{"90d":{"score":40,"rank":146,"coverage":0.55,"confidence":"Low"}},"description":"Distract-Bench evaluates robustness of vision-language models to semantic visual distractions, which are meaningful but task-irrelevant cues that preserve the ground-truth answer.","whyItMatters":"It exposes a distinct failure mode where models perceive evidence correctly but reason from distracting cues, shifting robustness evaluation from perceptual degradation to distraction handling.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3a746734efbdf9199a373fc70cc1f55de7066e6c16c76f7437758e552168964d"},"motivation":"Reasoning Vision-Language Models (VLMs) achieve strong performance on complex multimodal tasks, but reliable real-world application requires handling visual inputs that are messier than clean, curated benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08894","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Yizheng-Sun","organizationType":"community","sourceUrl":"https://github.com/Yizheng-Sun/Distract-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_dlawbench_4866c9b4","familyId":"bmf_48d9ddc30394","name":"DLawBench","oneLine":"DLawBench evaluates multi-turn legal consultation in Chinese and U.S. law across four client personas. It scores LLMs on information gathering, legal reasoning, and claim support using 461 cases, 5,532 paired fact entries, inquiry and issue rubrics, and a public leaderboard.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13931","pdf":"https://arxiv.org/pdf/2606.13931","project":null,"code":"https://github.com/SKYLENAGE-AI/DLawBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.13931"},"evidence":{"snippet":"To fill this gap, we introduce DLawBench, a diagnostic benchmark for real-world legal consultation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13931"},"ranking":{"90d":{"score":33,"rank":221,"coverage":0.55,"confidence":"Low"}},"description":"DLawBench evaluates multi-turn legal consultation in Chinese and U.S. law across four client personas. It scores LLMs on information gathering, legal reasoning, and claim support using 461 cases, 5,532 paired fact entries, inquiry and issue rubrics, and a public leaderboard.","whyItMatters":"Existing legal benchmarks assume complete fact patterns, overlooking the interactive elicitation needed in real consultations. DLawBench provides a diagnostic evaluation of LLM capability in legal consultation, revealing performance gaps and failure modes that inform development of models for legal assistance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2c69f1f9a9969aa88bb2f102fa252169159f7fd5b0cc104ee8b1b1d068085f0e"},"motivation":"Lawyer-client consultation is a critical starting point for legal services.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13931","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SKYLENAGE-AI","organizationType":"community","sourceUrl":"https://github.com/SKYLENAGE-AI/DLawBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_dmv-bench_d16bc24a","familyId":"bmf_bac3db190dd7","name":"DMV-Bench","oneLine":"DMV-Bench evaluates visual memory of multimodal agents in an interactive shopping environment with 1,000 product variants. Agents run autonomous sessions and must recall cued product images via exact URL match, with text leakage controlled.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Multimodal"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.27499","pdf":"https://arxiv.org/pdf/2606.27499","project":null,"code":"https://github.com/yyyujintang/DMV-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.27499"},"evidence":{"snippet":"We introduce DMV-Bench (Code: https://github.com/yyyujintang/DMV-Bench), the first interactive benchmark for multimodal-agent visual memory.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.27499"},"ranking":{"90d":{"score":27,"rank":286,"coverage":0.7,"confidence":"Medium"}},"description":"DMV-Bench evaluates visual memory of multimodal agents in an interactive shopping environment with 1,000 product variants. Agents run autonomous sessions and must recall cued product images via exact URL match, with text leakage controlled.","whyItMatters":"Existing agent memory benchmarks focus on text; DMV-Bench isolates visual memory needs in interactive settings, offering a controlled protocol for comparing agent architectures on pixel-based recall across varying session lengths.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4228d80f2293441b65e01027ef3ace973887d6a3a795b65dd5354248bcd0fee9"},"motivation":"Research on agent memory has matured rapidly, but almost entirely on the text side: few existing benchmarks ask, in an interactive environment, when an agent genuinely needs to remember what it saw rather than what it could write down.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.27499","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DMV-Bench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/yyyujintang/DMV-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_doc2ci_60d5a789","familyId":"bmf_8bd9fe5a8f8c","name":"Doc2CI","oneLine":"DOC2CI evaluates LLM-generated CI/CD configuration YAML against reference configurations from four CI services, measuring exact match and schema validity.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.01451","pdf":"https://arxiv.org/pdf/2608.01451","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01451"},"evidence":{"snippet":"We introduce DOC2CI, a benchmark of 3,363 description-to-YAML pairs collected from the official documentation of four CI services, and evaluate 14 open-weight models from 7B-34B parameters together with GPT-4o and GPT-4.1, producing over 53,000 configurations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01451"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DOC2CI evaluates LLM-generated CI/CD configuration YAML against reference configurations from four CI services, measuring exact match and schema validity.","whyItMatters":"It quantifies the gap between LLM-generated configuration validity and reference similarity, highlighting the need for schema-aware evaluation in configuration generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a0c3a19d7cff8e576878c0f419ebb72da9a1b1cd6be61ae444f8907390adc617"},"motivation":"Adopting Continuous Integration (CI) often requires writing YAML configurations that are error-prone and challenging to maintain.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.01451","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_doc2db-bench_2f955ada","familyId":"bmf_589f2dd26eba","name":"Doc2DB-Bench","oneLine":"Doc2DB-Bench evaluates document-to-database construction, converting long heterogeneous documents into normalized relational databases with entity identities, keys, cross-table links, and integrity constraints. It includes 203 document instances across 42 schemas and 7 domains, with fine-grained capability annotations for intra-table extraction and inter-table reasoning.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.08459","pdf":"https://arxiv.org/pdf/2608.08459","project":null,"code":"https://github.com/SetonLiang/Doc2DB-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.08459"},"evidence":{"snippet":"We introduce Doc2DB-Bench, a benchmark for Document-to-Database construction, containing 203 long-document instances across 42 schemas and seven domain groups, with 117 entity tables, 132 relationship tables, 7,341 rows, and 41,935 cells.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08459"},"ranking":{"30d":{"score":28,"rank":97,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":267,"coverage":0.55,"confidence":"Low"}},"description":"Doc2DB-Bench evaluates document-to-database construction, converting long heterogeneous documents into normalized relational databases with entity identities, keys, cross-table links, and integrity constraints. It includes 203 document instances across 42 schemas and 7 domains, with fine-grained capability annotations for intra-table extraction and inter-table reasoning.","whyItMatters":"Existing document-to-table benchmarks overlook relational database requirements such as normalization, entity resolution, and cross-table consistency. Doc2DB-Bench addresses this gap by providing a testbed for assessing LLM-based systems in realistic database construction, supporting analytics, compliance, and decision-making applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1a30e1a1267c2ee86ee536f5d0089979d7489042c158d140a7974e1c3b2d2840"},"motivation":"Practical AI systems increasingly need to turn long, heterogeneous documents into queryable relational databases, not isolated spreadsheets.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.08459","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Doc2DB-Bench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/SetonLiang/Doc2DB-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_docformbench_158426e5","familyId":"bmf_7b27e2f9f67b","name":"DocFormBench","oneLine":"DocFormBench evaluates content-aware document formatting for LLMs and multimodal models, using accuracy and efficiency metrics, with a workflow method DocFormFlow for target localization and modification.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.01936","pdf":"https://arxiv.org/pdf/2606.01936","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01936"},"evidence":{"snippet":"This content-aware setting remains challenging and underexplored, primarily due to the lack of dedicated evaluation datasets.To enable evaluation in realistic content-aware scenarios, we introduce DocFormBench, a benchmark that extends Text-to-Format evaluation to diverse formatting requirements, along with metrics for both accuracy and efficiency.To mitigate redundant document reading in existing methods during formatting, we propose DocFormFlow, a workflow formatting method that decouples targ","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01936"},"ranking":{},"description":"DocFormBench evaluates content-aware document formatting for LLMs and multimodal models, using accuracy and efficiency metrics, with a workflow method DocFormFlow for target localization and modification.","whyItMatters":"Existing formatting benchmarks lack content-aware evaluation, leaving a gap in assessing target identification. This benchmark addresses that by providing diverse formatting requirements and metrics for accuracy and efficiency, enabling practical comparison of formatting models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0920d8cee2885bea75469eb16ce61d085ba8678b37b7ce072f88a79cc65ce840"},"motivation":"Recent advances in large language models (LLMs) have opened up new possibilities for automated document formatting.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01936","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_docprivacybench_1e095c41","familyId":"bmf_db22e09b280e","name":"DocPrivacyBench","oneLine":"DocPrivacyBench evaluates susceptibility of document understanding MLLMs to relational privacy leakage when visual evidence is absent or minimal, using KIE tasks on identity documents.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12911","pdf":"https://arxiv.org/pdf/2608.12911","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12911"},"evidence":{"snippet":"It suppresses the leakage of high-risk field pairs while preserving KIE performance.Second, we introduce DocPrivacyBench, a novel benchmark to systematically evaluate a model's susceptibility to privacy leakage under conditions of absent or minimal visual evidence.Third, we evaluate three MLLMs and six unlearning methods using this benchmark, assessing both post-unlearning leakage suppression and utility preservation.Our results demonstrate that existing MLLMs consistently exhibit privacy leakag","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12911"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"DocPrivacyBench evaluates susceptibility of document understanding MLLMs to relational privacy leakage when visual evidence is absent or minimal, using KIE tasks on identity documents.","whyItMatters":"Addresses underexplored privacy vulnerabilities in document MLLMs, providing a framework to assess and mitigate leakage of correlated sensitive fields, which is crucial for trustworthy deployment in document processing.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b077a7d75b0671f82623b40664d07463f3e8457ba5233ff336ba1821edca71df"},"motivation":"While the privacy risks of multimodal large language models (MLLMs) have drawn significant attention, the unique vulnerabilities of domain-specific MLLMs remain largely underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12911","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_1df91e499cc9d53e","familyId":"catalog_family_1df91e499cc9d53e","name":"DocVQA","oneLine":"A dataset for Visual Question Answering on document images containing 50,000 questions defined on 12,000+ document images. The benchmark tests AI's ability to understand document structure and content, requiring models to comprehend document layout and perform information retrieval to answer questions about document images.","description":"A dataset for Visual Question Answering on document images containing 50,000 questions defined on 12,000+ document images. The benchmark tests AI's ability to understand document structure and content, requiring models to comprehend document layout and perform information retrieval to answer questions about document images.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Image To Text","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/docvqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1df91e499cc9d53e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/docvqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"docvqa","url":"https://llm-stats.com/benchmarks/docvqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image to text","multimodal","vision"],"catalogModelCount":28,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_42d39fa20c2f1413","familyId":"catalog_family_42d39fa20c2f1413","name":"DocVQAtest","oneLine":"DocVQA is a Visual Question Answering benchmark on document images containing 50,000 questions defined on 12,000+ document images. The benchmark focuses on understanding document structure and content to answer questions about various document types including letters, memos, notes, and reports from the UCSF Industry Documents Library.","description":"DocVQA is a Visual Question Answering benchmark on document images containing 50,000 questions defined on 12,000+ document images. The benchmark focuses on understanding document structure and content to answer questions about various document types including letters, memos, notes, and reports from the UCSF Industry Documents Library.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/docvqatest","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_42d39fa20c2f1413"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/docvqatest"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"docvqatest","url":"https://llm-stats.com/benchmarks/docvqatest","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","vision"],"catalogModelCount":11,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_dosebench_cf8cdec2","familyId":"bmf_828ba2bfddf1","name":"DOSEBENCH","oneLine":"DOSEBENCH evaluates LLM decision-making on over-the-counter dosing questions, with 81 curated scenarios for adult acetaminophen and ibuprofen use. Correct answers require tracking dose timing, computing rolling 24-hour intake, and following product-label constraints.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.04262","pdf":"https://arxiv.org/pdf/2606.04262","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04262"},"evidence":{"snippet":"We introduce DOSEBENCH, a focused benchmark of 81 curated OTC dosing scenarios focused on adult acetaminophen and ibuprofen use, with manually annotated gold references.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04262"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DOSEBENCH evaluates LLM decision-making on over-the-counter dosing questions, with 81 curated scenarios for adult acetaminophen and ibuprofen use. Correct answers require tracking dose timing, computing rolling 24-hour intake, and following product-label constraints.","whyItMatters":"Addresses an underexplored safety-relevant medical QA setting. Evaluates temporal reasoning, constraint following, and uncertainty handling, showing that confident responses can violate dosing constraints.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"35cb50ac28ed5751da38f6dc540086da39e58e9b4641d5d407186162205a041d"},"motivation":"Large language models (LLMs) are increasingly used for everyday health questions, including whether a user can safely take another dose of an over-the-counter (OTC) medication.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04262","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_755debb970fbb325","familyId":"catalog_family_755debb970fbb325","name":"Doubao Multi-Turn Bench","oneLine":"Doubao Multi-Turn Bench evaluates models on multi-turn conversational tasks, measuring context retention, instruction following, and coherent reasoning across extended dialogues.","description":"Doubao Multi-Turn Bench evaluates models on multi-turn conversational tasks, measuring context retention, instruction following, and coherent reasoning across extended dialogues.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Instruction Following","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/doubao-multi-turn-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_755debb970fbb325"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/doubao-multi-turn-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"doubao-multi-turn-bench","url":"https://llm-stats.com/benchmarks/doubao-multi-turn-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["instruction following","general","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_dr-cik_017f835a","familyId":"bmf_6f7e7a58db67","name":"Dr-CiK","oneLine":"Dr-CiK is a benchmark for evaluating whether agents can retrieve forecasting-relevant supporting context from a document corpus, filter distractors, distill evidence, and generate forecasts. It provides context ablations and evaluates deep research and forecasting methods.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27904","pdf":"https://arxiv.org/pdf/2605.27904","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27904"},"evidence":{"snippet":"Therefore, we introduce Dr-CiK, a benchmark for evaluating whether agents can retrieve forecasting-relevant supporting context from a document corpus, filter out distractors, distill the retrieved context into forecast-useful evidence, and generate forecasts supported by that evidence.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27904"},"ranking":{},"description":"Dr-CiK is a benchmark for evaluating whether agents can retrieve forecasting-relevant supporting context from a document corpus, filter distractors, distill evidence, and generate forecasts. It provides context ablations and evaluates deep research and forecasting methods.","whyItMatters":"Real-world forecasting requires active discovery of external context from noisy sources. Dr-CiK assesses the entire pipeline of context retrieval and use, revealing that most agents recover little evidence and are misled by distractors, guiding development of foresight-driven agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c646db2a54322b4c8e26f2be5bc60cf8e4b7041375f56757268633a575e86d29"},"motivation":"Time series forecasting in real-world settings often depends not only on historical observations, but also on external context that must be actively discovered from noisy, heterogeneous information sources.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27904","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_c8c3d2e4ed91e9d5","familyId":"catalog_family_c8c3d2e4ed91e9d5","name":"DRACO","oneLine":"DRACO is a deep research benchmark that evaluates an agent's ability to gather, synthesize, and reason over information to answer complex research questions. Scores are based on official rubrics per question, with the final score being the average across all questions.","description":"DRACO is a deep research benchmark that evaluates an agent's ability to gather, synthesize, and reason over information to answer complex research questions. Scores are based on official rubrics per question, with the final score being the average across all questions.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Reasoning","Search","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c8c3d2e4ed91e9d5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/draco"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/draco"}],"catalogSources":[{"catalog":"benchlm","sourceId":"draco","url":"https://benchlm.ai/benchmarks/draco","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Data Research and Analysis with Complex Operations","format":"Normalized rubric score","tasks":"Agentic data research and analysis tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"draco","url":"https://llm-stats.com/benchmarks/draco","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","search","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Search & Retrieval"],"domainScope":"general"},{"id":"bm_dragon_22a47a0d","familyId":"bmf_a9c43be948c5","name":"DragOn","oneLine":"DragOn is a drag grounding benchmark and training dataset for GUI agents across four interaction types: text highlighting, cell selection, element resizing, and slider manipulation, with 3.5M training tasks and 2,000 evaluation examples.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06322","pdf":"https://arxiv.org/pdf/2606.06322","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.06322"},"evidence":{"snippet":"We introduce DragOn, a drag grounding benchmark and training dataset covering four domains: text highlighting, cell selection, element resizing and slider manipulation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06322"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DragOn is a drag grounding benchmark and training dataset for GUI agents across four interaction types: text highlighting, cell selection, element resizing, and slider manipulation, with 3.5M training tasks and 2,000 evaluation examples.","whyItMatters":"Drag-based interactions are under-represented in existing datasets. DragOn provides a large-scale resource for improving drag grounding, which may enhance performance on downstream computer-use tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"84df10f69fd73f8ae9d74ca51c0066c685f26b761e5ffd7cb9a9d22284c3c662"},"motivation":"GUI agents - vision-based models that control desktops, web browsers, and mobile devices through graphical user interfaces - promise to automate a wide range of digital tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"publication_reported","venue":"Published as a workshop paper at SCALE - 43rd International Conference on Machine Learning, Seoul, South Korea. PMLR 306, 2026","evidence":"Published as a workshop paper at SCALE - 43rd International Conference on Machine Learning, Seoul, South Korea. PMLR 306, 2026","evidenceUrl":"https://arxiv.org/abs/2606.06322","source":"arxiv-journal-reference","evidenceLevel":"strong-author-metadata","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publications":[{"venueName":"Published as a workshop paper at SCALE - 43rd International Conference on Machine Learning, Seoul, South Korea. PMLR 306, 2026","publicationStatus":"published","evidence":[{"sourceType":"arxiv-journal-reference","sourceUrl":"https://arxiv.org/abs/2606.06322","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Published as a workshop paper at SCALE - 43rd International Conference on Machine Learning, Seoul, South Korea. PMLR 306, 2026","level":"strong-author-metadata"}]}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_drawingvqa_70f98873","familyId":"bmf_869e9acc7076","name":"DrawingVQA","oneLine":"The evaluation object is unclear from the provided information.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.15418","pdf":"https://arxiv.org/pdf/2607.15418","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.15418"},"evidence":{"snippet":"We introduce DrawingVQA, the first benchmark designed to evaluate multimodal large language models (MLLMs) on real-world construction drawings -- a core media in architecture, civil, and many other engineering practices.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.15418"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"The evaluation object is unclear from the provided information.","whyItMatters":"The evaluation gap and practical decision value are unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"df73c25747555150e087c24786463c9155a6c10100f2f7b19dc88f17c4c0701c"},"motivation":"We introduce DrawingVQA, the first benchmark designed to evaluate multimodal large language models (MLLMs) on real-world construction drawings -- a core media in architecture, civil, and many other engineering practices.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"paper","evidence":"CVPR 2026 Findings accepted paper","evidenceUrl":"https://arxiv.org/abs/2607.15418","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"paper","reviewStatus":"accepted","decisionRaw":"CVPR 2026 Findings accepted paper","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.15418","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"CVPR 2026 Findings accepted paper","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_drflow_2f7bee63","familyId":"bmf_c4ede3eb8785","name":"DRFLOW","oneLine":"DRFLOW evaluates an agent's ability to predict personalized workflows, sequences of action-steps, from heterogeneous sources across five domains. It includes 100 tasks with reference workflow steps and multiple diagnostic metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18191","pdf":"https://arxiv.org/pdf/2606.18191","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18191"},"evidence":{"snippet":"Therefore, we introduce DRFLOW, a benchmark for evaluating personalized workflows predicted by agents from heterogeneous sources.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18191"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"DRFLOW evaluates an agent's ability to predict personalized workflows, sequences of action-steps, from heterogeneous sources across five domains. It includes 100 tasks with reference workflow steps and multiple diagnostic metrics.","whyItMatters":"Deep research systems are typically evaluated on report generation, but enterprise tasks often require actionable workflows. DRFLOW addresses this gap by assessing workflow prediction, offering metrics for factual grounding, step recovery, and personalization, which can guide development of more practical agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"508b75b001ccec7896909217fa938ce1b4edaef5dec2911a20804095a69eed10"},"motivation":"Deep research (DR) systems are increasingly used for complex information-seeking tasks, but existing works mainly focus on generating reports and summaries.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18191","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_drgbt-1k_b65fb238","familyId":"bmf_71bb9565fcb3","name":"DRGBT-1K","oneLine":"DRGBT-1K is a large-scale benchmark for dynamic RGBT tracking with 1,045 real-world sequences, 795K frame pairs, dense annotations, and a unified evaluation protocol across 20 trackers, plus an online leaderboard.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19772","pdf":"https://arxiv.org/pdf/2607.19772","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19772"},"evidence":{"snippet":"5) We develop an online evaluation platform for DRGBT-1K and provide a leaderboard that collects all methods evaluated on this benchmark.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19772"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DRGBT-1K is a large-scale benchmark for dynamic RGBT tracking with 1,045 real-world sequences, 795K frame pairs, dense annotations, and a unified evaluation protocol across 20 trackers, plus an online leaderboard.","whyItMatters":"Dynamic modality and platform variations are underrepresented in tracking benchmarks; this provides systematic evaluation for robustness under real-world transitions, supporting tracker comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d36fb4d1380e516196afe7d479cb5c559c8828ccc7ef02fe1326b96164528943"},"motivation":"Dynamic RGBT (DRGBT) tracking aims to continuously localize a target when the available sensing modalities and observation platforms vary over time.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19772","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_drinq_46ba6b80","familyId":"bmf_b8e493ef53e0","name":"DRInQ","oneLine":"The evaluation targets conversational implicature in question utterances, using a semi-automated pipeline to generate question-context-interpretation instances with controlled variation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24267","pdf":"https://arxiv.org/pdf/2605.24267","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24267"},"evidence":{"snippet":"We introduce DRinQ, a benchmark for evaluating pragmatic reasoning about conversational implicature in question utterances, designed to isolate pragmatic variation while holding each question's surface form fixed.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24267"},"ranking":{},"description":"The evaluation targets conversational implicature in question utterances, using a semi-automated pipeline to generate question-context-interpretation instances with controlled variation.","whyItMatters":"This evaluation probes the gap between generation and inference in pragmatic reasoning, but no public comparison path is provided beyond the paper's findings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b11a5d2686f1106926a2e82766ebf4db4fb0c795ea44823b5010389f0f204c61"},"motivation":"Human conversation relies heavily on conversational implicature, in which speakers convey meanings that are suggested rather than explicitly stated.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24267","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_drivespatial_f218db55","familyId":"bmf_f22a23f585f9","name":"DRIVESPATIAL","oneLine":"DriveSpatial is a benchmark of 15.6K human-verified QA pairs across 20 tasks from five AD datasets, evaluating VLMs on cognitive scene construction, multi-view relational understanding, temporal reasoning, and generalization.","area":"Vision & 3D","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Automotive"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.23176","pdf":"https://arxiv.org/pdf/2605.23176","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.23176"},"evidence":{"snippet":"We introduce DriveSpatial, a benchmark of 15.6K human-verified QA pairs across 20 tasks from five large-scale AD datasets.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.23176"},"ranking":{},"description":"DriveSpatial is a benchmark of 15.6K human-verified QA pairs across 20 tasks from five AD datasets, evaluating VLMs on cognitive scene construction, multi-view relational understanding, temporal reasoning, and generalization.","whyItMatters":"Existing AD vision-language benchmarks focus on static, single-view QA, leaving unclear whether VLMs can reason over dynamic driving scenes. DriveSpatial provides a multi-sourced, human-verified protocol to measure spatiotemporal reasoning and reveals a substantial human-model gap.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0a5d90c49b179b9c945fd98805ee2f9c3e4a362e7f16159a1657981f60a0f746"},"motivation":"Spatiotemporal intelligence in autonomous driving (AD) requires an agent to integrate multi-view observations into a coherent scene representation, maintain object continuity across viewpoints and time, and reason about spatial relations, interactions, and future dynamics.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.23176","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_d90ee9ccf6bea1d2","familyId":"catalog_family_d90ee9ccf6bea1d2","name":"DROP","oneLine":"DROP (Discrete Reasoning Over Paragraphs) is a reading comprehension benchmark requiring discrete reasoning over paragraph content. It contains crowdsourced, adversarially-created questions that require resolving references and performing discrete operations like addition, counting, or sorting, demanding comprehensive paragraph understanding beyond paraphrase-and-entity-typing shortcuts.","description":"DROP (Discrete Reasoning Over Paragraphs) is a reading comprehension benchmark requiring discrete reasoning over paragraph content. It contains crowdsourced, adversarially-created questions that require resolving references and performing discrete operations like addition, counting, or sorting, demanding comprehensive paragraph understanding beyond paraphrase-and-entity-typing shortcuts.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d90ee9ccf6bea1d2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/drop"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/drop"}],"catalogSources":[{"catalog":"benchlm","sourceId":"drop","url":"https://benchlm.ai/benchmarks/drop","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"Discrete Reasoning Over Paragraphs","format":"F1","tasks":"Paragraph reasoning questions","successorKey":null},{"catalog":"llm-stats","sourceId":"drop","url":"https://llm-stats.com/benchmarks/drop","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","math"],"catalogModelCount":30,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_drugbench_dc75a792","familyId":"bmf_e6b83865905a","name":"DrugBench","oneLine":"DrugBench evaluates AI control protocols for mitigating medication-related harm using 3,671 medical conversations and FDA drug labels, covering drug interactions, contraindications, dosing constraints, and patient action restrictions. Introduces severity-based monitoring.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-10","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.20663","pdf":"https://arxiv.org/pdf/2606.20663","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.20663"},"evidence":{"snippet":"To this end, we introduce DrugBench, an AI control evaluation benchmark which combines 3,671 multi-turn medical conversations from HealthBench with drug information from official FDA labels, covering four categories of medication-related harm: drug interactions, contraindications, dosing constraints, and patient action restrictions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.20663"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DrugBench evaluates AI control protocols for mitigating medication-related harm using 3,671 medical conversations and FDA drug labels, covering drug interactions, contraindications, dosing constraints, and patient action restrictions. Introduces severity-based monitoring.","whyItMatters":"Addresses the safety-critical need to evaluate external safeguards for LLMs in medical QA, beyond simple accuracy. Useful for developers of safe medical AI systems and control protocols.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"175b15a623efd90db47501eae3eb494c23dd635f8d448b835c1d81e7434ba9e6"},"motivation":"Large Language Models have the potential to expand and improve the access to clinical information by enabling new ways of interacting with medical knowledge in natural language.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.20663","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_bb19a3a75070caab","familyId":"catalog_family_bb19a3a75070caab","name":"DS-Arena-Code","oneLine":"Data Science Arena Code benchmark for evaluating LLMs on realistic data science code generation tasks. Tests capabilities in complex data processing, analysis, and programming across popular Python libraries used in data science workflows.","description":"Data Science Arena Code benchmark for evaluating LLMs on realistic data science code generation tasks. Tests capabilities in complex data processing, analysis, and programming across popular Python libraries used in data science workflows.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ds-arena-code","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bb19a3a75070caab"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ds-arena-code"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ds-arena-code","url":"https://llm-stats.com/benchmarks/ds-arena-code","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_7296791033c44d09","familyId":"catalog_family_7296791033c44d09","name":"DS-FIM-Eval","oneLine":"DeepSeek's internal Fill-in-the-Middle evaluation dataset for measuring code completion performance improvements in data science contexts","description":"DeepSeek's internal Fill-in-the-Middle evaluation dataset for measuring code completion performance improvements in data science contexts","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ds-fim-eval","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7296791033c44d09"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ds-fim-eval"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ds-fim-eval","url":"https://llm-stats.com/benchmarks/ds-fim-eval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_dsagentbench_96674f61","familyId":"bmf_3b3386cca544","name":"DSAgentBench","oneLine":"DSAgentBench evaluates language agents on end-to-end data-science workflows in real computer environments. It comprises 275 tasks spanning data wrangling, exploration, modeling, visualization, and validation, with deterministic evaluators verifying analytical correctness, visual outputs, and model performance.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.10366","pdf":"https://arxiv.org/pdf/2608.10366","project":null,"code":"https://github.com/vis-nlp/DSAgentBench","data":null,"hfPaper":"https://huggingface.co/papers/2608.10366"},"evidence":{"snippet":"We introduce DSAgentBench, the first benchmark to evaluate whether agents can automate full data-science workflows inside real computer environments.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":10,"hfDailySubmittedAt":"2026-08-12T00:00:00.000Z","githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10366"},"ranking":{"30d":{"score":32,"rank":71,"coverage":0.85,"confidence":"High"},"90d":{"score":32,"rank":222,"coverage":0.7,"confidence":"Medium"}},"description":"DSAgentBench evaluates language agents on end-to-end data-science workflows in real computer environments. It comprises 275 tasks spanning data wrangling, exploration, modeling, visualization, and validation, with deterministic evaluators verifying analytical correctness, visual outputs, and model performance.","whyItMatters":"Existing benchmarks lack real-computer interaction and fail to capture the multi-stage, multi-tool nature of data-science practice. DSAgentBench provides a realistic environment for assessing whether agents can automate complete workflows, highlighting a significant capability gap and offering a foundation for developing grounded, verifiable autonomous agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"13a48bd9c8ed99ab691727a6ba073787497b8648c31139f631ca072d3c8ca472"},"motivation":"Real-world data science involves long-horizon workflows that span data wrangling, exploration, modeling, visualization, and validation, and require coordinated use of tools such as notebooks, IDEs, terminals, browsers, and databases within real operating environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10366","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"vis-nlp","organizationType":"academic-lab","sourceUrl":"https://github.com/vis-nlp/DSAgentBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_ef6f49085728d89e","familyId":"catalog_family_ef6f49085728d89e","name":"DSBench-FullStack","oneLine":"DSBench-FullStack is DeepSeek's internal full-stack development test set for evaluating coding agents on end-to-end software engineering tasks.","description":"DSBench-FullStack is DeepSeek's internal full-stack development test set for evaluating coding agents on end-to-end software engineering tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://api-docs.deepseek.com/zh-cn/updates/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ef6f49085728d89e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/dsbenchfullstack"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/dsbench-fullstack"}],"catalogSources":[{"catalog":"benchlm","sourceId":"dsBenchFullStack","url":"https://benchlm.ai/benchmarks/dsbenchfullstack","paperUrl":"https://api-docs.deepseek.com/zh-cn/updates/","year":"2026","fullName":"DeepSeek DSBench FullStack","format":"Provider-reported score","tasks":"Internal full-stack coding-agent tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"dsbench-fullstack","url":"https://llm-stats.com/benchmarks/dsbench-fullstack","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","agents","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_a72275bd6c7025a9","familyId":"catalog_family_a72275bd6c7025a9","name":"DSBench-Hard","oneLine":"DSBench-Hard is DeepSeek's internal test set of difficult coding-agent problems.","description":"DSBench-Hard is DeepSeek's internal test set of difficult coding-agent problems.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://api-docs.deepseek.com/zh-cn/updates/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a72275bd6c7025a9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/dsbenchhard"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/dsbench-hard"}],"catalogSources":[{"catalog":"benchlm","sourceId":"dsBenchHard","url":"https://benchlm.ai/benchmarks/dsbenchhard","paperUrl":"https://api-docs.deepseek.com/zh-cn/updates/","year":"2026","fullName":"DeepSeek DSBench Hard","format":"Provider-reported score","tasks":"Internal hard coding-agent tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"dsbench-hard","url":"https://llm-stats.com/benchmarks/dsbench-hard","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","agents","code"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_60e77101e76950a9","familyId":"catalog_family_60e77101e76950a9","name":"DUDE","oneLine":"DUDE (Document Understanding Dataset and Evaluation) tests multi-page, multi-domain document understanding and reasoning.","description":"DUDE (Document Understanding Dataset and Evaluation) tests multi-page, multi-domain document understanding and reasoning.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/dude","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_60e77101e76950a9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/dude"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"dude","url":"https://llm-stats.com/benchmarks/dude","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","multimodal","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_dumatebench_7a085138","familyId":"bmf_546d4c680ff6","name":"DuMateBench","oneLine":"Autonomous agents are increasingly adopted to complete complex, multi-tool workflows in real-world settings.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.26546","pdf":"https://arxiv.org/pdf/2608.26546","project":"https://dumatebench.com/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.26546"},"evidence":{"snippet":"We introduce DuMateBench, a real-session benchmark reconstructed from anonymized and privacy-screened user sessions collected from a large-scale production agent platform.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26546"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"Autonomous agents are increasingly adopted to complete complex, multi-tool workflows in real-world settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.26546","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_dungeonbench_a497b825","familyId":"bmf_d20232b4c917","name":"DungeonBench","oneLine":"DungeonBench evaluates tactical reasoning in Dungeons & Dragons combat with two tracks: Encounter for single fights and Day for linked encounters with persistent resources, using a shared decision stream of complete tactical observations and legal options.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.29577","pdf":"https://arxiv.org/pdf/2607.29577","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.29577"},"evidence":{"snippet":"We introduce DungeonBench, a benchmark for tactical reasoning in Dungeons & Dragons combat, built to cover the vast majority of combat-relevant 2014 System Reference Document content whose effects can be resolved by the simulator while retaining mechanics that simplified combat simulators often abstract away.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.29577"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DungeonBench evaluates tactical reasoning in Dungeons & Dragons combat with two tracks: Encounter for single fights and Day for linked encounters with persistent resources, using a shared decision stream of complete tactical observations and legal options.","whyItMatters":"Current benchmarks often under-test rules-rich tactical reasoning where geometry, timing, resources, and rule interactions matter. DungeonBench fills this gap by providing a reproducible simulator-based environment with clear scoring, allowing comparison of policies from heuristic controllers to language models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f2f2c45dc95d1fedc1b91d2dfe45ef30f1361711c32bbf8a32f74cad93d009ca"},"motivation":"Games and simulators make valuable benchmarks by turning decisions into measurable outcomes, but many current suites under-test rules-rich tactical reasoning: the ability to choose well when geometry, timing, resources, objectives, and rule interactions all matter at once.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.29577","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DungeonBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.29577","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_dunphybench_f4ebbce5","familyId":"bmf_4789198ca55d","name":"DunphyBench","oneLine":"DunphyBench evaluates long-horizon embodied decision-making in housing environments, requiring agents to navigate and choose options aligned with multi-dimensional human preferences under partial observations.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Agents","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.01456","pdf":"https://arxiv.org/pdf/2608.01456","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01456"},"evidence":{"snippet":"In this work, we propose DunphyBench, a new benchmark for evaluating agents on long-horizon human-centered embodied decision-making, where the agent must navigate through multiple embodied housing environments and make decisions that align with multi-dimensional human preferences.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01456"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DunphyBench evaluates long-horizon embodied decision-making in housing environments, requiring agents to navigate and choose options aligned with multi-dimensional human preferences under partial observations.","whyItMatters":"Addresses the gap in evaluating agents on long-horizon, human-centered decisions beyond procedural tasks, providing a reference for progress in integrating multimodal evidence and preference reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"52cbb7e90341a443df6a2c1eaf28568c944b39cf56792cab39a5f5d383eccc9e"},"motivation":"Agents are increasingly expected to act not only as task executors, but also as decision-makers on behalf of human users.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.01456","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_duobench_6c56fbaf","familyId":"bmf_bd016e251030","name":"DuoBench","oneLine":"DuoBench is a benchmarking framework for bimanual manipulation policies on the FR3 Duo platform, with eleven tasks across four coordination categories, simulated and partially real-world, plus human-teleoperated datasets. It proposes stage-based evaluation beyond binary success.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11901","pdf":"https://arxiv.org/pdf/2606.11901","project":"https://duobench.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.11901"},"evidence":{"snippet":"We introduce DuoBench, an extensible benchmarking framework for bimanual manipulation policies on the FR3 Duo platform.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.11901"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"DuoBench is a benchmarking framework for bimanual manipulation policies on the FR3 Duo platform, with eleven tasks across four coordination categories, simulated and partially real-world, plus human-teleoperated datasets. It proposes stage-based evaluation beyond binary success.","whyItMatters":"Provides a reproducible testbed for diagnosing failures in dual-arm policy learning, covering coordination challenges not captured by existing benchmarks. Useful for robotics researchers working on bimanual manipulation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"91b39f668a6ea3e7139246b4a6f3a126742af454aa3488105b80f5fc8faa4fc8"},"motivation":"Bimanual robot systems substantially expand manipulation capabilities, but coordinating two arms introduces additional control complexity and failure modes that are not well captured by existing benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11901","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_dynabench_70c2d7da","familyId":"bmf_124c02b8d60a","name":"DynaBench","oneLine":"A benchmark for robot manipulation in dynamic environments, including dynamic manipulation and bimanual coordination tasks.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Agents","Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.22119","pdf":"https://arxiv.org/pdf/2607.22119","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.22119"},"evidence":{"snippet":"To rigorously evaluate these capabilities, we introduce DynaBench, a novel benchmark for robot manipulation in dynamic environments.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.22119"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark for robot manipulation in dynamic environments, including dynamic manipulation and bimanual coordination tasks.","whyItMatters":"Evaluates sample-efficient multi-agent cooperation in dynamic settings, addressing causal limitations in multi-stream policies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0f96e13d92c5264485c7e8464a18c491cda9ec15d4b6ae64e089f255718710cb"},"motivation":"Multi-stream robot manipulation policies achieve unparalleled sample efficiency and generalization by modeling actions relative to environmental reference frames.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.22119","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_ab9fe765bfe10432","familyId":"catalog_family_ab9fe765bfe10432","name":"DynaMath","oneLine":"A multimodal mathematics and reasoning benchmark focused on dynamic visual problem solving.","description":"A multimodal mathematics and reasoning benchmark focused on dynamic visual problem solving.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Math","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ab9fe765bfe10432"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/dynamath"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/dynamath"}],"catalogSources":[{"catalog":"benchlm","sourceId":"dynaMath","url":"https://benchlm.ai/benchmarks/dynamath","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"DynaMath","format":"Multimodal mathematical reasoning","tasks":"Dynamic visual math problems","successorKey":null},{"catalog":"llm-stats","sourceId":"dynamath","url":"https://llm-stats.com/benchmarks/dynamath","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","math","multimodal","reasoning","vision"],"catalogModelCount":7,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_dynamicmem_faf51af5","familyId":"bmf_bbff9299325a","name":"DynamicMem","oneLine":"DynamicMem evaluates long-horizon memory in LLM agents through a synthetic benchmark with 15 months of multi-app activity per user across 16 applications, including attributes, habits, and preferences that evolve and must be inferred from scattered evidence. Scoring occurs at quarterly checkpoints, tracking performance as history grows.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22877","pdf":"https://arxiv.org/pdf/2606.22877","project":"https://wenyaxie023.github.io/DynamicMem/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.22877"},"evidence":{"snippet":"We introduce DynamicMem, a synthetic benchmark that constructs 15 months of activity per user, providing long-term multi-app data that real users' privacy keeps out of reach.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22877"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"DynamicMem evaluates long-horizon memory in LLM agents through a synthetic benchmark with 15 months of multi-app activity per user across 16 applications, including attributes, habits, and preferences that evolve and must be inferred from scattered evidence. Scoring occurs at quarterly checkpoints, tracking performance as history grows.","whyItMatters":"Existing memory benchmarks use short, simplified interactions, missing real-world complexity. DynamicMem provides a long-horizon, multi-app evaluation that yields insights into memory failures, such as degradation with history length and retrieval-driven errors, which can guide improvements in memory systems for personal assistants.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a12059d7f5c24799fcc4e6e7d2f7faa87292b2a9544de5f4f2694b6e6459213a"},"motivation":"LLM agents increasingly act as personal assistants that must remember a user's profile over months: who they are (attributes), what they routinely do (habits), and what they prefer (preferences), and keep it updated as jobs, routines, and tastes drift.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22877","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_e-bench_c1793d08","familyId":"bmf_e19c9d617150","name":"E-Bench","oneLine":"E-Bench evaluates multi-step tool-use agents in synthetic state-changing tasks across three product domains: Honor of Kings, QQ Music, and Tencent Meeting. It requires agents to discover hidden information and compose multiple tool calls before changing state, with deterministic grading by database-state diffs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.23722","pdf":"https://arxiv.org/pdf/2607.23722","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.23722"},"evidence":{"snippet":"We introduce E-Bench, a fully synthetic benchmark with 323 state-changing tasks across three product domains: Honor of Kings, QQ Music, and Tencent Meeting.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23722"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"E-Bench evaluates multi-step tool-use agents in synthetic state-changing tasks across three product domains: Honor of Kings, QQ Music, and Tencent Meeting. It requires agents to discover hidden information and compose multiple tool calls before changing state, with deterministic grading by database-state diffs.","whyItMatters":"E-Bench addresses the gap in evaluating complex tool-use agents that interact with stateful environments over multiple steps, providing a scalable and controllable alternative to existing benchmarks that often focus on isolated API calls or short trajectories.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b14b055828bb69da9e040feab6fc201b2d5d920996f91ee438b70cc1a5dd7e2b"},"motivation":"Large Language Models (LLMs) are increasingly deployed as agents that interact with stateful environments over multiple steps: gathering hidden information, composing tool calls, and committing state changes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.23722","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_earlydx_668e5f94","familyId":"bmf_8ad7d2d6cad1","name":"EarlyDx","oneLine":"EarlyDx is a benchmark for open-ended early diagnosis from emergency department encounters in MIMIC-IV, using admission-time records and LLM-auditor-supervised free-text labels.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.28788","pdf":"https://arxiv.org/pdf/2607.28788","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.28788"},"evidence":{"snippet":"We introduce EarlyDx, a large-scale benchmark for open-ended early diagnosis, built from 154,834 emergency department encounters in MIMIC-IV.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.28788"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EarlyDx is a benchmark for open-ended early diagnosis from emergency department encounters in MIMIC-IV, using admission-time records and LLM-auditor-supervised free-text labels.","whyItMatters":"It addresses evaluation of diagnosis prediction under realistic admission constraints, which existing closed-set benchmarks miss.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cdf7b32c5e6afa906bb2eaca16c826fdf1ad6a12315b6799b6f505e0b0dcd34c"},"motivation":"Clinical diagnosis at hospital admission must be made rapidly from limited, incomplete evidence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.28788","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_earnings25_937fc49b","familyId":"bmf_c5e53bcda5ed","name":"Earnings25","oneLine":"Earnings25 is a finance-domain benchmark for evaluating automatic speech recognition (ASR) on English-language earnings calls. It includes testset-full (498 hours of S&P 500 earnings calls from Q4 2025) and testset-segmented (46 hours of 290 segments from 2025 U.S. earnings calls), with aligned transcripts and metadata like speaker roles and industry labels, enabling speaker- and industry-aware evaluation beyond WER.","area":"Speech & Audio","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-07-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.23813","pdf":"https://arxiv.org/pdf/2607.23813","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.23813"},"evidence":{"snippet":"We introduce Earnings25, a finance-domain benchmark for evaluating automatic speech recognition (ASR) on English-language earnings calls under realistic conditions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23813"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Earnings25 is a finance-domain benchmark for evaluating automatic speech recognition (ASR) on English-language earnings calls. It includes testset-full (498 hours of S&P 500 earnings calls from Q4 2025) and testset-segmented (46 hours of 290 segments from 2025 U.S. earnings calls), with aligned transcripts and metadata like speaker roles and industry labels, enabling speaker- and industry-aware evaluation beyond WER.","whyItMatters":"Existing ASR benchmarks lack domain-specific, large-scale, and realistic finance data, and typically report only aggregate word error rate. Earnings25 provides a reproducible protocol and structured metadata to support more granular evaluation of ASR systems in the finance domain.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e37fc638ca9da3a66fa076f8aa9dabfc3e80673d78ab7a1750908710a039dc6f"},"motivation":"We introduce Earnings25, a finance-domain benchmark for evaluating automatic speech recognition (ASR) on English-language earnings calls under realistic conditions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.23813","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Earnings25 team","organizationType":"benchmark-organization","sourceUrl":"https://arxiv.org/abs/2607.23813","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_earthverse_1bd65be4","familyId":"bmf_90c3f7c1b48c","name":"EarthVerse","oneLine":"Evaluates scientific agents on 405 reproducible tasks across 199 documented natural hazard events, scoring fine-grained answer units and task-specific rubrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.23525v1","pdf":"https://arxiv.org/pdf/2608.23525v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce EarthVerse, a benchmark that evaluates scientific agents through package-scoped investigations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23525"},"ranking":{"30d":{"score":40,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates scientific agents on 405 reproducible tasks across 199 documented natural hazard events, scoring fine-grained answer units and task-specific rubrics.","whyItMatters":"Provides a rigorous, provenance-focused measure of end-to-end scientific reliability for agents dealing with dynamic Earth systems and natural hazards.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"48fc702768bc377acbcfbb50cfc38f769bfdf088ca2ac499fab70351e2938f3a"},"motivation":"Earth-system analysis reconstructs changing physical processes from observations that differ in source, scale, timing, and modality.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark includes executable ground truth, rubrics, and evaluation of 25 systems, with a reproducible basis stated, meeting criteria for a published benchmark.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce EarthVerse, a benchmark that evaluates scientific agents through package-scoped investigations."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.23525v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:42:57.448509Z"},"attentionForecast":{"score":58,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses scientific agent evaluation with a large task suite and public evaluation setup, likely to interest both AI and Earth science communities."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_paint-what-you-see-benchmarking-dexterous-_f2ccb455","familyId":"bmf_95ac01e427df","name":"EASEL","oneLine":"Evaluates dexterous visual tool use through reference-guided painting, semantic annotation, handwriting, and path planning tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Tool use"],"topics":["Agents","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-26","firstSeenAt":"2026-08-27","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.25417","pdf":"https://arxiv.org/pdf/2608.25417","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We propose EASEL, a benchmark evaluating a controlled instance of dexterous visual tool use that adopts reference-guided visual reconstruction as its primary proxy task: the agent incrementally paints a canvas to match a reference image.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25417"},"ranking":{"30d":{"score":41,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":45,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates dexterous visual tool use through reference-guided painting, semantic annotation, handwriting, and path planning tasks.","whyItMatters":"Introduces closed-loop, parameterized visual action as an underexplored agent capability beyond static QA and navigation.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"219bb05fc88581bb086f2d83b839cddcd213bffe5c0821a73a21f7e5af520ab1"},"motivation":"Evaluation is shifting from static QA toward agentic settings where models act through external tools.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:12:36.511123Z","model":"deepseek-v4-pro","decisionReason":"EASEL is explicitly proposed as a benchmark with dataset and model contributions; the paper reports evaluation of 25 models, and artifact links are absent but likely forthcoming.","canonicalNameSource":"abstract","canonicalNameEvidence":"We propose EASEL, a benchmark evaluating a controlled instance of dexterous visual tool use that adopts reference-guided visual reconstruction as its primary proxy task"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25417","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":70,"confidence":"Low","horizon":"7d","reason":"Broad multimodal agent topic with large-scale data and multiple model evaluations, though no public code link is provided."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Agents","Tool Calling","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_ebench_35561bef","familyId":"bmf_e19c9d617150","name":"EBench","oneLine":"EBench is a simulation benchmark for diagnosing generalist mobile manipulation policies. It comprises 26 manipulation tasks annotated along five capability dimensions (scene, atomic skill, horizon, precision, mobility) and four generalization dimensions (object, background, instruction, mixed), with strict train/test splits and held-out online evaluation.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18239","pdf":"https://arxiv.org/pdf/2606.18239","project":null,"code":"https://github.com/InternRobotics/EBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.18239"},"evidence":{"snippet":"We present EBench, a simulation benchmark that diagnoses generalist mobile manipulation policies beyond a single success-rate scalar.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":16,"hfDailySubmittedAt":"2026-06-25T00:00:00.000Z","githubStars":132,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18239"},"ranking":{"90d":{"score":59,"rank":21,"coverage":0.7,"confidence":"Medium"}},"description":"EBench is a simulation benchmark for diagnosing generalist mobile manipulation policies. It comprises 26 manipulation tasks annotated along five capability dimensions (scene, atomic skill, horizon, precision, mobility) and four generalization dimensions (object, background, instruction, mixed), with strict train/test splits and held-out online evaluation.","whyItMatters":"EBench addresses the need for multi-axis diagnostic evaluation of generalist manipulation models, moving beyond single success-rate metrics. It provides practical decision value by revealing capability and generalization profiles that guide model iteration and selection.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2b36e4ea16e26ac2f96a4b76eeeb920cb7cc763be748ef428da62af36d77d8c5"},"motivation":"We present EBench, a simulation benchmark that diagnoses generalist mobile manipulation policies beyond a single success-rate scalar.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18239","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Shanghai AI Laboratory","organizationType":"company-research-lab","sourceUrl":"https://github.com/InternRobotics/EBench","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_ec-reason-bench_6994eec4","familyId":"bmf_e8fbcd70b60f","name":"EC-Reason-Bench","oneLine":"EC-Reason-Bench is a training-free diagnostic protocol for LLM enzyme classification, isolating four reasoning levers: output structure, external knowledge, reasoning structure, and robustness, to measure why LLMs fail on EC number prediction.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning","Factuality"],"topics":["Reasoning"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.26397","pdf":"https://arxiv.org/pdf/2607.26397","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.26397"},"evidence":{"snippet":"We propose EC-Reason-Bench, a training-free, diagnostic evaluation protocol built to answer two questions: why general LLMs score close to nothing on EC number prediction, and how much of that loss can be recovered without updating a single weight.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26397"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EC-Reason-Bench is a training-free diagnostic protocol for LLM enzyme classification, isolating four reasoning levers: output structure, external knowledge, reasoning structure, and robustness, to measure why LLMs fail on EC number prediction.","whyItMatters":"The protocol reveals that external knowledge is decisive and that reasoning acts as an arbiter among conflicting neighbors, showing that single-number leaderboards obscure the source of accuracy loss. This informs how to design knowledge-integration methods for protein function prediction.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"60aac0fe5509c09669f763909ee7d0f84bd843e63f84d7e5e3df9a9f64038552"},"motivation":"Enzyme function prediction is a hierarchical, knowledge-intensive form of protein function classification.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26397","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_ecg-interpbench_df0017be","familyId":"bmf_d1b9cc173fbe","name":"ECG-InterpBench","oneLine":"ECG-InterpBench evaluates interpretability of ECG foundation models using sparse autoencoders with matched capacity, measuring reconstruction fidelity, clinical feature accessibility, and cross-seed reproducibility across 450 cells.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.27404","pdf":"https://arxiv.org/pdf/2607.27404","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.27404"},"evidence":{"snippet":"We introduce ECG-InterpBench, a benchmark designed to systematically evaluate the interpretability of ECG foundation-model representations.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27404"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ECG-InterpBench evaluates interpretability of ECG foundation models using sparse autoencoders with matched capacity, measuring reconstruction fidelity, clinical feature accessibility, and cross-seed reproducibility across 450 cells.","whyItMatters":"Performance-focused ECG benchmarks ignore whether representations are interpretable. This benchmark provides a controlled, reproducible framework for comparing models on interpretability, aiding clinical adoption where understanding model decisions is critical.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4a3801e94abcd9d28a5d01f9e8b61e4468f7f25e4274beccea887421e3dac8f7"},"motivation":"Existing benchmarks for electrocardiogram foundation models primarily evaluate downstream predictive performance, providing limited insight into whether their internal representations can be faithfully decomposed, clinically interpreted, or reproduced across independent analyses.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27404","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_b17fe74e48c6cabd","familyId":"catalog_family_b17fe74e48c6cabd","name":"ECLeKTic","oneLine":"A multilingual closed-book question answering dataset that evaluates cross-lingual knowledge transfer in large language models across 12 languages, using knowledge-seeking questions based on Wikipedia articles that exist only in one language","description":"A multilingual closed-book question answering dataset that evaluates cross-lingual knowledge transfer in large language models across 12 languages, using knowledge-seeking questions based on Wikipedia articles that exist only in one language","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/eclektic","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b17fe74e48c6cabd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/eclektic"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"eclektic","url":"https://llm-stats.com/benchmarks/eclektic","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ecoagent-bench_99f855a3","familyId":"bmf_5ccd0648d9b6","name":"EcoAgent-Bench","oneLine":"EcoAgent-Bench evaluates LLM agents' economic decision-making under budget constraints across 304 tasks in five families, with priced actions and metrics for micro accuracy and economic consistency.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.05519","pdf":"https://arxiv.org/pdf/2608.05519","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.05519"},"evidence":{"snippet":"We introduce EcoAgent-Bench, in which every task specifies priced actions and an explicit budget.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.05519"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EcoAgent-Bench evaluates LLM agents' economic decision-making under budget constraints across 304 tasks in five families, with priced actions and metrics for micro accuracy and economic consistency.","whyItMatters":"Existing agent benchmarks measure task completion without considering cost-effectiveness. EcoAgent-Bench explicitly tests the trade-off between completion and resource use, providing a distinct evaluation dimension.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c3fe677923cbaccb685dda4d50f22346ef6e985d82cec5774787a6cd75fca190"},"motivation":"Agent benchmarks usually measure task completion and treat resource use as an auxiliary statistic.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.05519","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ecomagentbench_d7694649","familyId":"bmf_7733964f2cda","name":"EComAgentBench","oneLine":"EComAgentBench evaluates LLM-based shopping agents on 662 long-horizon product selection tasks built from real Amazon data. Each task requires uncovering hidden requirements spread across an explicit query, tool-gated profile, and scripted clarification, verifying candidates, and committing to one product within 100 tool calls. Scoring uses typed, source-tagged rubrics for each requirement.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.17698","pdf":"https://arxiv.org/pdf/2606.17698","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.17698"},"evidence":{"snippet":"To address this gap, we introduce EComAgentBench, a benchmark of 662 tasks grounded in real Amazon products and reviews.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.17698"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EComAgentBench evaluates LLM-based shopping agents on 662 long-horizon product selection tasks built from real Amazon data. Each task requires uncovering hidden requirements spread across an explicit query, tool-gated profile, and scripted clarification, verifying candidates, and committing to one product within 100 tool calls. Scoring uses typed, source-tagged rubrics for each requirement.","whyItMatters":"Existing shopping benchmarks reveal full intent upfront, failing to reflect real-world requirements that emerge over time. EComAgentBench measures agents' ability to handle long-horizon interactions, providing a reproducible foundation for comparing dependable shopping assistance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c4d3fc0fd3c1488e9eda11c19090312cb3837eb2a1ce41843f9c0eb38eed9261"},"motivation":"As LLM-based shopping agents enter production, existing benchmarks fail to capture how a shopper's requirements arrive: stated implicitly in the query, recorded in a profile, or revealed only when the right question is asked.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.17698","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_edgebench_88dc3c92","familyId":"bmf_2c4d59d4e34e","name":"EdgeBench","oneLine":"EdgeBench evaluates autonomous agents on 134 real-world tasks across scientific discovery, software engineering, optimization, knowledge work, formal mathematics, and games. Each task requires 12+ hours of continuous interaction with multi-level feedback. Scoring tracks agent performance over time (at 2,4,6,8,10,12 hours). A public leaderboard is maintained; 51 tasks and evaluation framework are open-sourced.","area":"Agents & Tool Use","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":["Environment learning","Long-horizon task execution","Learning from feedback"],"topics":["Agents","Environment Learning","Scaling Laws"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.05155","pdf":"https://arxiv.org/pdf/2607.05155","project":"https://edge-bench.org/","code":"https://github.com/ByteDance-Seed/EdgeBench","data":"https://huggingface.co/datasets/ByteDance-Seed/EdgeBench","hfPaper":"https://huggingface.co/papers/2607.05155"},"evidence":{"snippet":"This discovery stems from EdgeBench, a suite of 134 real world tasks with ultra-long horizons, spanning scientific discovery, software engineering, combinatorial optimization, professional knowledge work, formal mathematics, and interactive games. […] We publicly release 51 tasks and our full evaluation framework to accelerate the study of how agents learn from real world experience.","reasonCodes":["coined title prefix ending in Bench or Benchmark","named benchmark identity and task release stated in nearby sentences","evaluation protocol evidence"]},"dataStatus":"primary-source-reviewed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":18,"hfDailySubmittedAt":"2026-07-07T00:00:00.000Z","githubStars":432,"githubScope":"benchmark_repo","hfDatasetDownloads":8042,"hfDatasetLikes":83},"source":{"type":"arxiv","id":"2607.05155"},"ranking":{"90d":{"score":76,"rank":4,"coverage":1.0,"confidence":"High","datasetDownloadRank":5,"datasetRankPopulation":66}},"description":"EdgeBench evaluates autonomous agents on 134 real-world tasks across scientific discovery, software engineering, optimization, knowledge work, formal mathematics, and games. Each task requires 12+ hours of continuous interaction with multi-level feedback. Scoring tracks agent performance over time (at 2,4,6,8,10,12 hours). A public leaderboard is maintained; 51 tasks and evaluation framework are open-sourced.","whyItMatters":"EdgeBench fills a gap in evaluating agents' ability to learn from real-world environments over extended periods, providing a standardized protocol for comparing long-horizon learning capabilities. It offers a public leaderboard and open-source tasks, enabling reproducible comparisons and tracking of agent improvement over time, which is valuable for model selection and development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bf947e2af808593cdcd52d4e4578c40a22f3628308d6650745e5a65db4ffa6b3"},"motivation":"Pretraining scaling laws reveal that model capability improves predictably with data and compute.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"source-reviewed","reviewedAt":"2026-08-19","sources":["https://arxiv.org/abs/2607.05155","https://edge-bench.org/","https://github.com/ByteDance-Seed/EdgeBench","https://huggingface.co/datasets/ByteDance-Seed/EdgeBench"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.05155","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"releaseDates":{"firstPublicAt":"2026-07-02","paperV1At":"2026-07-06"},"publishers":[{"name":"ByteDance Seed","organizationType":"company-research-lab","sourceUrl":"https://github.com/ByteDance-Seed/EdgeBench","role":"benchmark-publisher"}],"capabilityGroups":["Agents","Coding & Software Engineering","Mathematics & Formal Sciences"],"domainScope":"specific","catalogSources":[{"catalog":"benchlm","sourceId":"edgeBench","url":"https://benchlm.ai/benchmarks/edgebench","paperUrl":"https://edge-bench.org/","year":"2026","fullName":"EdgeBench","format":"Time-budgeted agent learning curves","tasks":"Systems and software-engineering tasks","successorKey":null},{"catalog":"benchlm","sourceId":"edgeBench","url":"https://benchlm.ai/benchmarks/edgebench","paperUrl":"https://edge-bench.org/paper.pdf","year":"2026","fullName":"EdgeBench","format":"Long-horizon interactive agent evaluation","tasks":"134 tasks (51 public) across 6 domains","successorKey":null}],"catalogCategories":["coding","agentic"],"catalogModelCount":0,"catalogStarCount":0},{"id":"bm_edit2tikz_4730ba41","familyId":"bmf_95a10c6e66b2","name":"Edit2TikZ","oneLine":"Edit2TikZ evaluates instruction-guided scientific figure editing with TikZ code, featuring 1,548 samples with textual or visual localization requests and multi-step edits, using a human-aligned evaluation framework to measure edit completion and content preservation.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.13441","pdf":"https://arxiv.org/pdf/2608.13441","project":null,"code":"https://github.com/Solunny/Edit2TikZ","data":null,"hfPaper":"https://huggingface.co/papers/2608.13441"},"evidence":{"snippet":"We introduce Edit2TikZ, a comprehensive benchmark for scientific figure editing tasks, featuring 1,548 diverse and high-quality samples.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13441"},"ranking":{"30d":{"score":23,"rank":152,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":356,"coverage":0.55,"confidence":"Low"}},"description":"Edit2TikZ evaluates instruction-guided scientific figure editing with TikZ code, featuring 1,548 samples with textual or visual localization requests and multi-step edits, using a human-aligned evaluation framework to measure edit completion and content preservation.","whyItMatters":"Existing benchmarks focus on figure reconstruction or generation, leaving a gap for systematic evaluation of instruction-guided editing with compilable code. This benchmark provides a standardized protocol for assessing models' ability to perform precise, code-based edits while preserving unrelated content, offering practical value for developing reliable multimodal systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c72b462be79393ac1f33e98bd8b2621eb4ba990f93da0280006cf195e2551067"},"motivation":"Although multimodal large language models (MLLMs) have shown substantial potential in visual understanding and graphic code generation, editing scientific figures through code presents a greater challenge: a model must jointly recover visual structure, ground the requested change, generate compilable code, and preserve all unrelated content.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13441","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Solunny","organizationType":"community","sourceUrl":"https://github.com/Solunny/Edit2TikZ","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_editclevr_a686a8c4","familyId":"bmf_5225fce0cba4","name":"EditCLEVR","oneLine":"EditCLEVR is a paired-scene intervention benchmark for object-centric representations, with before/after CLEVR renders and a known attribute change. Includes metrics like SGIA and Delta-SGIA for semantic faithfulness, with probe-free diagnostics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-19","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.22705","pdf":"https://arxiv.org/pdf/2607.22705","project":null,"code":"https://github.com/torux-bughunter/EditCLEVR","data":null,"hfPaper":"https://huggingface.co/papers/2607.22705"},"evidence":{"snippet":"We introduce EditCLEVR, a paired-scene intervention benchmark in which each example contains a before/after pair of CLEVR-style renders with the same object indices and scene layout, and either exactly one known attribute change on one known object or a no-edit re-render for drift measurement.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.22705"},"ranking":{"90d":{"score":23,"rank":369,"coverage":0.55,"confidence":"Low"}},"description":"EditCLEVR is a paired-scene intervention benchmark for object-centric representations, with before/after CLEVR renders and a known attribute change. Includes metrics like SGIA and Delta-SGIA for semantic faithfulness, with probe-free diagnostics.","whyItMatters":"Provides a direct test of whether per-object representations behave correctly under controlled semantic edits, addressing a gap in evaluating compositional faithfulness beyond segmentation or single-image prediction.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"af332deae070394600265404b352ea94ee159983b428e0171ed4517e722d4687"},"motivation":"Object-centric learning aims to represent scenes as objects whose properties can be reused in new combinations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.22705","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"torux-bughunter","organizationType":"community","sourceUrl":"https://github.com/torux-bughunter/EditCLEVR","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_eduart_6e0cd321","familyId":"bmf_b8da88c300f6","name":"EduArt","oneLine":"EduArt evaluates art-historical knowledge and visual reasoning in multimodal LLMs using 871 human-authored questions in Italian and English, covering multiple formats and languages. Scoring is based on accuracy and psychometric properties.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Factuality"],"topics":["Multimodal","Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.02007","pdf":"https://arxiv.org/pdf/2607.02007","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.02007"},"evidence":{"snippet":"This paper introduces EduArt, an educational-level benchmark for art-historical knowledge and visual reasoning in multimodal LLMs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02007"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EduArt evaluates art-historical knowledge and visual reasoning in multimodal LLMs using 871 human-authored questions in Italian and English, covering multiple formats and languages. Scoring is based on accuracy and psychometric properties.","whyItMatters":"General benchmarks don't reveal discipline-specific capabilities. EduArt provides a fine-grained benchmark to assess art historical knowledge, showing that format significantly affects performance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"dbb51aece7ec6c9f44b723106dfcb5e8c97989d5a966442965c740fbd4fb9ffc"},"motivation":"Large language models now score near ceiling on general benchmarks, but these aggregate measures reveal little about how models behave within single disciplines.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02007","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_educlaw-bench_d4a7a280","familyId":"bmf_9d60a04e1c2d","name":"EduClaw-Bench","oneLine":"EduClaw-Bench evaluates pedagogical LLM agents in a simulated 30-day tutoring relationship with a knowledge-tracing-based simulated learner, scoring learning gain, responsiveness, helpfulness, and curriculum design across 55 scenarios.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03206","pdf":"https://arxiv.org/pdf/2608.03206","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03206"},"evidence":{"snippet":"We introduce EduClaw-Bench, a benchmark that places an agent tutor in a continuous 30-day relationship with a simulated learner grounded in knowledge tracing (KT), whose knowledge-concept mastery, from a KT model trained on real-student data, drives its answers and is probed for learning gain across 55 scenarios.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03206"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EduClaw-Bench evaluates pedagogical LLM agents in a simulated 30-day tutoring relationship with a knowledge-tracing-based simulated learner, scoring learning gain, responsiveness, helpfulness, and curriculum design across 55 scenarios.","whyItMatters":"Existing benchmarks focus on single-turn tasks, leaving long-horizon tutoring unmeasured. This benchmark provides a standardized way to assess sustained pedagogical interaction, helping developers and educators choose agents that maintain effective teaching over time.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a7b659278f29aba313d02677f5db56cb5adb8ef8455ff8b3d47b1f457ab078fb"},"motivation":"Large language models (LLMs) power educational applications from tutoring to essay scoring, but each is a point solution to a single task, and only recently have these point solutions been integrated into agents operating over a learning management system (LMS).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03206","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_edupluginbench_5240df80","familyId":"bmf_74fa30a0270a","name":"EduPluginBench","oneLine":"EduPluginBench is an executable benchmark for evaluating code-generation models on producing plugins that meet governed ecosystem requirements including least privilege, telemetry consent, provenance, and bounded failure. It uses 1,440 mutants and 120 clean references with staged admission levels P0-P4.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-01","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.00739","pdf":"https://arxiv.org/pdf/2608.00739","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00739"},"evidence":{"snippet":"We introduce EduPluginBench, an executable benchmark and staged admission method for generated plugins in governed software ecosystems.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00739"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EduPluginBench is an executable benchmark for evaluating code-generation models on producing plugins that meet governed ecosystem requirements including least privilege, telemetry consent, provenance, and bounded failure. It uses 1,440 mutants and 120 clean references with staged admission levels P0-P4.","whyItMatters":"Compilation and functional tests do not ensure compliance with security and governance constraints. EduPluginBench provides a staged admission method targeting these gaps, enabling assessment of generated plugins in governed environments, which is critical for safe deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2932025763fd783563ed3470cceefc30734d2540086e0a2ef320354e6012e47e"},"motivation":"Code-generation models can produce executable components, but compilation and functional tests do not establish compliance with least privilege, telemetry consent, provenance, privileged-write authority, lifecycle constraints, or bounded failure.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00739","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_eduvideobench_97f1a2a9","familyId":"bmf_3b433f3c988d","name":"EduVideoBench","oneLine":"Benchmark for educational validity of video generation models, grounded in the Knowledge-Skills-Attitude framework.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.26918","pdf":"https://arxiv.org/pdf/2605.26918","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.26918"},"evidence":{"snippet":"In this work, we present EduVideoBench, the first balanced benchmark in the education domain, grounded in the Knowledge-Skills-Attitude (KSA) framework so that pedagogical adequacy and educational safety are evaluated jointly rather than as ad-hoc quality dimensions.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26918"},"ranking":{},"description":"Benchmark for educational validity of video generation models, grounded in the Knowledge-Skills-Attitude framework.","whyItMatters":"Existing video benchmarks ignore pedagogical validity, which is critical for classroom use. EduVideoBench fills this gap.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6b1bcd356847d462da4565c563b63171ce4a7d471f2ced9d2e91acd0170e9849"},"motivation":"Video generation models (VGMs) are rapidly entering classrooms, yet existing benchmarks evaluate only perceptual quality, intrinsic faithfulness, generic safety, or video as a reasoning medium, and none assesses whether the outputs are educationally valid.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26918","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_eeg-editbench_8f55ec4b","familyId":"bmf_23b204771077","name":"EEG-EditBench","oneLine":"EEG-EditBench evaluates EEG-to-image retrieval models using 2,137 controlled edits of 200 THINGS-EEG2 test images, covering object identity, attributes, background, and presence, with metrics like 200-way accuracy and 2AFC accuracy.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.27857","pdf":"https://arxiv.org/pdf/2607.27857","project":null,"code":"https://github.com/XiaoZhangYES/EEG-EditBench","data":"https://huggingface.co/datasets/xiaozgg/EEG-EditBench","hfPaper":"https://huggingface.co/papers/2607.27857"},"evidence":{"snippet":"Motivated by this question, we introduce EEG-EditBench, a diagnostic benchmark that examines this question through controlled edits of object identity, attributes, background, and object presence.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":859,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2607.27857"},"ranking":{"90d":{"score":29,"rank":248,"coverage":0.85,"confidence":"High","datasetDownloadRank":18,"datasetRankPopulation":66}},"description":"EEG-EditBench evaluates EEG-to-image retrieval models using 2,137 controlled edits of 200 THINGS-EEG2 test images, covering object identity, attributes, background, and presence, with metrics like 200-way accuracy and 2AFC accuracy.","whyItMatters":"Standard retrieval accuracy can mask whether models truly preserve visual information. EEG-EditBench provides a controlled, repeatable protocol for probing fine-grained visual distinctions, informing development of more robust EEG decoding models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"424af334db84c36047b41a0ae784bbd71965283c112866750ab2cca8e4125c80"},"motivation":"Recent EEG-to-image retrieval models have achieved strong performance in identifying viewed images from semantically diverse candidates.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27857","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Search & Retrieval"],"domainScope":"general"},{"id":"bm_ego-mc-bench_91c3532a","familyId":"bmf_4df235e7cb24","name":"Ego-MC-Bench","oneLine":"Ego-MC-Bench is a benchmark for evaluating reactive step-by-step task guidance in cooking scenarios, focusing on timely interventions when mistakes occur.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09547","pdf":"https://arxiv.org/pdf/2606.09547","project":"https://apratimbh.github.io/livecookv2/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.09547"},"evidence":{"snippet":"To evaluate this crucial capability, we introduce Ego-MC-Bench (Mistake Corrections), a benchmark for evaluating reactive, step-by-step task guidance in realistic cooking scenarios.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09547"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Ego-MC-Bench is a benchmark for evaluating reactive step-by-step task guidance in cooking scenarios, focusing on timely interventions when mistakes occur.","whyItMatters":"The benchmark addresses the capability of video LLMs to intervene proactively during task execution, which is crucial for practical guidance assistants.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"efe3fea054f05e12cc57d5c8a88984974ac3199414ebe5a629dea5eda6950954"},"motivation":"Learning everyday skills, like cooking a dish, relies increasingly on instructional media such as online videos.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09547","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_ego-metas_beb4f202","familyId":"bmf_9a895d99bf07","name":"Ego-METAS","oneLine":"Ego-METAS evaluates online temporal action segmentation in egocentric video, where models must select sensor modalities (RGB, audio, gaze, IMU, monochrome) per timestep to maximize accuracy under hardware-representative energy budgets. It includes 100+ hours of untrimmed video from multiple datasets.","area":"Multimodal","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02246","pdf":"https://arxiv.org/pdf/2606.02246","project":"https://maria-sanvil.github.io/Ego-METAS-website/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02246"},"evidence":{"snippet":"To address this, we introduce Ego-METAS: the first Egocentric online Multimodal Energy-efficient Temporal Action Segmentation benchmark.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02246"},"ranking":{},"description":"Ego-METAS evaluates online temporal action segmentation in egocentric video, where models must select sensor modalities (RGB, audio, gaze, IMU, monochrome) per timestep to maximize accuracy under hardware-representative energy budgets. It includes 100+ hours of untrimmed video from multiple datasets.","whyItMatters":"This benchmark addresses the gap in energy-aware perception for embodied AI by providing a standardized testbed for developing and comparing cost-aware sensor routing policies in continuous, untrimmed environments. It enables assessment of trade-offs between predictive accuracy and energy consumption, with practical implications for always-on devices.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9a24f5d4789ce7e589fc3d7a39ddabcb0363a4dc8651cd70203d5c79eea65d13"},"motivation":"To operate in the physical world, embodied agents must perceive their environment in an \"always-on\" fashion, selectively accessing the most informative sensors to balance energy constraints and task accuracy.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02246","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Ego-METAS team","organizationType":"academic-lab","sourceUrl":"https://maria-sanvil.github.io/Ego-METAS-website/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_egoafford_be5c2b17","familyId":"bmf_7252a0a76b4f","name":"EgoAfford","oneLine":"EgoAfford is a benchmark for egocentric referring segmentation with task-oriented affordance grounding, comprising 15.5k images and 102 real images. It includes EgoLens, a 3B MLLM reference model.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04533","pdf":"https://arxiv.org/pdf/2608.04533","project":"https://egoafford.github.io","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.04533"},"evidence":{"snippet":"We introduce EgoAfford, a benchmark designed to connect these three aspects.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04533"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EgoAfford is a benchmark for egocentric referring segmentation with task-oriented affordance grounding, comprising 15.5k images and 102 real images. It includes EgoLens, a 3B MLLM reference model.","whyItMatters":"It addresses the need for connecting perception and planning in tabletop tasks, but the lack of public artifacts and scoring details limits its immediate utility.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d83b9224f5f13677b55e6ee0c150748cb38e0571156e38b0723803cd61d93ea4"},"motivation":"Part-level affordance grounding has advanced the localization of functional object regions associated with elemental actions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04533","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_egobench_d0f8580c","familyId":"bmf_31eeb47a8e16","name":"EgoBench","oneLine":"EgoBench is an interactive multimodal benchmark for tool-using agents. It comprises 1,045 egocentric-video-grounded tasks across four daily scenarios, with a user-agent-tool interactive environment. It assesses multimodal perception, tool invocation with multi-hop reasoning, and dynamic user interaction.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27820","pdf":"https://arxiv.org/pdf/2605.27820","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27820"},"evidence":{"snippet":"To bridge this gap, we introduce EgoBench, the first interactive multimodal benchmark for tool-using agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27820"},"ranking":{},"description":"EgoBench is an interactive multimodal benchmark for tool-using agents. It comprises 1,045 egocentric-video-grounded tasks across four daily scenarios, with a user-agent-tool interactive environment. It assesses multimodal perception, tool invocation with multi-hop reasoning, and dynamic user interaction.","whyItMatters":"AI agents in open environments need joint multimodal and tool-use capabilities. EgoBench provides a standardized interactive environment and deterministic joint validation to objectively measure these skills, addressing a lack of comparable benchmarks for dynamic tool-using agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"51ad36a0df77af5d2ff7bfe561f6ac08591c213b89d6f85fee1d5df9e6633fd5"},"motivation":"As AI agents increasingly operate in open, real-world environments, they require a deep synergy of multimodal perception, tool invocation with multi-hop reasoning, and dynamic interaction with users.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27820","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_egogapbench_95dd2183","familyId":"bmf_db4933aac951","name":"EgoGapBench","oneLine":"EgoGapBench evaluates egocentric action selection in multi-agent scenes, isolating the ability to choose actions from the agent's perspective when other agents are present. The benchmark includes training and test splits with human performance as reference.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.00547","pdf":"https://arxiv.org/pdf/2607.00547","project":null,"code":"https://github.com/jhCOR/EgoGapBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.00547"},"evidence":{"snippet":"To isolate egocentric perspective understanding, we introduce EgoGapBench, a diagnostic benchmark for measuring action selection in multi-agent egocentric scenes.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.00547"},"ranking":{"90d":{"score":23,"rank":376,"coverage":0.55,"confidence":"Low"}},"description":"EgoGapBench evaluates egocentric action selection in multi-agent scenes, isolating the ability to choose actions from the agent's perspective when other agents are present. The benchmark includes training and test splits with human performance as reference.","whyItMatters":"Existing egocentric benchmarks conflate first-person view processing with perspective-taking, making it hard to isolate perspective understanding. EgoGapBench fills this gap by providing a controlled evaluation for a capability that is crucial for embodied AI and human-robot interaction, showing that state-of-the-art models fail at this task.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"406666ebd38ef0445abfd398a8a28da2642cd2c92a5c7e54796d1d4ff9fdd3c3"},"motivation":"Existing egocentric benchmarks have primarily constructed the egocentric setting from first-person-view data, which makes it difficult to evaluate egocentric perspective itself in isolation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.00547","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"jhCOR","organizationType":"academic-lab","sourceUrl":"https://github.com/jhCOR/EgoGapBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_egomonth_17f19a53","familyId":"bmf_dfcc09f0ca92","name":"EgoMonth","oneLine":"EgoMonth benchmarks month-level egocentric video understanding with 300+ hours from 20 participants over 20-120 days, 1,443 QA pairs, and a 14-task framework across schema consolidation, episodic indexing, and cascading reasoning.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.13113","pdf":"https://arxiv.org/pdf/2608.13113","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.13113"},"evidence":{"snippet":"We introduce EgoMonth, the first month-level egocentric video understanding benchmark.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13113"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"EgoMonth benchmarks month-level egocentric video understanding with 300+ hours from 20 participants over 20-120 days, 1,443 QA pairs, and a 14-task framework across schema consolidation, episodic indexing, and cascading reasoning.","whyItMatters":"Existing long-video benchmarks lack inter-clip spatiotemporal continuity, so they cannot assess memory across days or weeks. EgoMonth provides a temporal-grounded evaluation for long-term memory in MLLMs, revealing that even top models remain far below human performance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c972f9445040cd30ad8721a8f6f2cf010362aaef364b8d13be0f7c5b0677817e"},"motivation":"Recent advances in Multimodal Large Language Models (MLLMs) have led to substantial progress in video understanding, accompanied by a growing number of long video benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13113","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_egoprox_628390c5","familyId":"bmf_c4115f8c172a","name":"EgoProx","oneLine":"EgoProx evaluates multimodal large language models on egocentric 3D proximity reasoning, with tasks organized along a cognitive hierarchy covering intention, exploration, exploitation, and chain-of-actions reasoning.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Geometric reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24456","pdf":"https://arxiv.org/pdf/2605.24456","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24456"},"evidence":{"snippet":"To this end, we introduce EgoProx, a benchmark for egocentric 3D proximity reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24456"},"ranking":{},"description":"EgoProx evaluates multimodal large language models on egocentric 3D proximity reasoning, with tasks organized along a cognitive hierarchy covering intention, exploration, exploitation, and chain-of-actions reasoning.","whyItMatters":"It addresses the gap in assessing embodied 3D spatial reasoning in MLLMs, a capability important for robotics and augmented reality applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0dd961362af71913f44d54eaf7cad948ed6e418826fcf3159123ab8a92688ad7"},"motivation":"Humans constantly reason about 3D proximity, the relations between their body and surrounding objects, to guide perception and action in daily life.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"CVPR 2026","evidence":"Accepted to CVPR 2026","evidenceUrl":"https://arxiv.org/abs/2605.24456","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"CVPR 2026","reviewStatus":"accepted","decisionRaw":"Accepted to CVPR 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2605.24456","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to CVPR 2026","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_egosafe-bench_9e855ed0","familyId":"bmf_5dba8ef98ef1","name":"EgoSafe-Bench","oneLine":"EgoSafe-Bench evaluates visual safety understanding in first-person video, using 12,000 QA samples from 3,000 clips under the Hierarchical Reasoning Evaluation (HRE) protocol, which requires reasoning from feature anchoring to intent inference.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Safety","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.26518","pdf":"https://arxiv.org/pdf/2607.26518","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.26518"},"evidence":{"snippet":"To address this, we introduce EgoSafe-Bench, a benchmark specifically designed to probe forensic reasoning in egocentric safety scenarios.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26518"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"EgoSafe-Bench evaluates visual safety understanding in first-person video, using 12,000 QA samples from 3,000 clips under the Hierarchical Reasoning Evaluation (HRE) protocol, which requires reasoning from feature anchoring to intent inference.","whyItMatters":"Existing safety benchmarks rely on third-person footage and binary metrics, missing the causal reasoning gap in egocentric perception. EgoSafe-Bench provides a reusable protocol to assess whether LVLMs can move beyond correlation to forensic logic, informing model selection for safety-critical applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d4b06f9e2989d606d8b5fdfd974e1cc00cade31017738af31986054bd065a625"},"motivation":"Reliable visual safety understanding in real-world scenarios demands more than just object recognition; it requires causal reasoning under epistemic uncertainty.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26518","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_egosafetybench_4d21ada3","familyId":"bmf_4c45392c5812","name":"EgoSafetyBench","oneLine":"EgoSafetyBench is a diagnostic egocentric video benchmark of 1,200 robot-view scenarios to evaluate vision-language models as runtime safety guards. It assesses situational awareness across routine, suspicious, obvious, and contextual hazards, and visual-channel robustness against misleading in-scene text.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.00218","pdf":"https://arxiv.org/pdf/2607.00218","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.00218"},"evidence":{"snippet":"We introduce EgoSafetyBench, an egocentric video benchmark of 1,200 robot-view scenarios annotated at half-second granularity, to evaluate VLMs as streaming guards across two tracks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.00218"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"EgoSafetyBench is a diagnostic egocentric video benchmark of 1,200 robot-view scenarios to evaluate vision-language models as runtime safety guards. It assesses situational awareness across routine, suspicious, obvious, and contextual hazards, and visual-channel robustness against misleading in-scene text.","whyItMatters":"This benchmark addresses the practical need for safety guards that distinguish genuine hazards from superficially alarming but benign actions, and highlights the vulnerability to misleading signage, which can inform safer deployment of embodied VLMs in real-world settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"98c489eb2d4216f5b04646e759a36801531ad2874cb0d51bf52f0fc32cd8c5a9"},"motivation":"Vision-language models (VLMs) are now proposed as runtime safety guards for embodied agents in homes and factories.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.00218","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_egosat_1b88fd52","familyId":"bmf_710c21ef2296","name":"EgoSAT","oneLine":"EgoSAT evaluates vision-language models on egocentric video reasoning in streaming settings. It contains 1,997 videos (165 hours) and about 4,800 QA pairs covering retrospective, online, and prospective reasoning tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24422","pdf":"https://arxiv.org/pdf/2606.24422","project":"https://leiyj23.github.io/EgoSAT/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.24422"},"evidence":{"snippet":"We introduce EgoSAT, the first comprehensive benchmark for egocentric video reasoning in streaming settings, designed to evaluate the capabilities of modern vision-language models (VLMs).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24422"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"EgoSAT evaluates vision-language models on egocentric video reasoning in streaming settings. It contains 1,997 videos (165 hours) and about 4,800 QA pairs covering retrospective, online, and prospective reasoning tasks.","whyItMatters":"Serves as a unified benchmark for streaming egocentric interaction understanding, enabling assessment of temporal reasoning and confidence calibration in VLMs, which is crucial for reliable deployment in real-time applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6a329f1bd677c01702fb8c5b6626912e87069c971f8da4966f3953260f71da1c"},"motivation":"We introduce EgoSAT, the first comprehensive benchmark for egocentric video reasoning in streaming settings, designed to evaluate the capabilities of modern vision-language models (VLMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026","evidence":"Accepted to ECCV 2026. Project page: https://leiyj23.github.io/EgoSAT/","evidenceUrl":"https://arxiv.org/abs/2606.24422","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026","reviewStatus":"accepted","decisionRaw":"Accepted to ECCV 2026. Project page: https://leiyj23.github.io/EgoSAT/","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.24422","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to ECCV 2026. Project page: https://leiyj23.github.io/EgoSAT/","level":"author-claim"}]}],"publishers":[{"name":"EgoSAT Project","organizationType":"academic-lab","sourceUrl":"https://leiyj23.github.io/EgoSAT/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_65396ab97dcda4af","familyId":"catalog_family_65396ab97dcda4af","name":"EgoSchema","oneLine":"A diagnostic benchmark for very long-form video language understanding consisting of over 5000 human curated multiple choice questions based on 3-minute video clips from Ego4D, covering a broad range of natural human activities and behaviors","description":"A diagnostic benchmark for very long-form video language understanding consisting of over 5000 human curated multiple choice questions based on 3-minute video clips from Ego4D, covering a broad range of natural human activities and behaviors","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/egoschema","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_65396ab97dcda4af"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/egoschema"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"egoschema","url":"https://llm-stats.com/benchmarks/egoschema","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","vision"],"catalogModelCount":9,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_egostream_a12d21a7","familyId":"bmf_0c3b61ab28b8","name":"EGOSTREAM","oneLine":"Egostream evaluates streaming episodic memory in egocentric vision through 2,250 curated questions across seven cognitive dimensions, expanded into 8,528 recall-conditioned evaluations using the Answer Validity Window to test recall from instant to ultra-long-term.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.31557","pdf":"https://arxiv.org/pdf/2605.31557","project":"https://saroo25.github.io/Egostream/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.31557"},"evidence":{"snippet":"We introduce Egostream, a diagnostic benchmark for streaming episodic memory evaluation in egocentric vision.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.31557"},"ranking":{},"description":"Egostream evaluates streaming episodic memory in egocentric vision through 2,250 curated questions across seven cognitive dimensions, expanded into 8,528 recall-conditioned evaluations using the Answer Validity Window to test recall from instant to ultra-long-term.","whyItMatters":"Existing streaming video benchmarks lack fine-grained diagnosis of memory retention over time. Egostream provides a controlled protocol that separates genuine forgetting from world-state changes, enabling precise assessment of model memory capabilities across different temporal scales.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5121561deac1f12231b29a0c829c0d07caf161ee0bf0f209877b195f3580c403"},"motivation":"Continuous episodic memory is a core capability for autonomous agents operating in dynamic, real-world environments, yet current streaming video benchmarks provide limited tools for diagnosing what models remember and for how long.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.31557","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Egostream project team","organizationType":"academic-lab","sourceUrl":"https://saroo25.github.io/Egostream/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_egotactile_e93c424a","familyId":"bmf_ffce4b8dd699","name":"EgoTactile","oneLine":"EgoTactile is a dataset pairing egocentric video with full-hand pressure supervision for everyday objects, including a bare-hand transfer subset.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09243","pdf":"https://arxiv.org/pdf/2606.09243","project":"https://egotactile.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.09243"},"evidence":{"snippet":"Therefore, we introduce EgoTactile, a benchmark pairing egocentric video with full-hand pressure supervision for diverse everyday objects, incorporating a bare-hand transfer subset to enable generalization to natural scenarios.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09243"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"EgoTactile is a dataset pairing egocentric video with full-hand pressure supervision for everyday objects, including a bare-hand transfer subset.","whyItMatters":"The dataset supports research in estimating grasp pressure from egocentric video, which is relevant for VR and robotic manipulation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a2c31baf1b64c5e1fec7f2cc17285fdc77d831cc43a87b15a55cb585f15d8fba"},"motivation":"Estimating full-hand grasp pressure from egocentric video is critical for immersive VR and robotic manipulation, yet dense tactile sensing often relies on intrusive hardware.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML2026 spotlight","evidence":"Accepted to ICML2026 spotlight","evidenceUrl":"https://arxiv.org/abs/2606.09243","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML2026 spotlight","reviewStatus":"accepted","decisionRaw":"Accepted to ICML2026 spotlight","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.09243","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to ICML2026 spotlight","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_ehr-complex_08f2ee7c","familyId":"bmf_68a1189baa1d","name":"EHR-Complex","oneLine":"EHR-Complex evaluates clinical agent performance on interactive reasoning over MIMIC-IV electronic health records. It consists of about 52K tasks across six clinical intents, requiring agents to execute SQL or Python in a sandboxed environment to answer patient- and population-level queries. Scoring is based on exact-match accuracy against expected outcomes.","area":"Agents & Tool Use","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.23301","pdf":"https://arxiv.org/pdf/2606.23301","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.23301"},"evidence":{"snippet":"In this work, we introduce EHR-Complex, a large-scale benchmark designed for interactive clinical database reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.23301"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EHR-Complex evaluates clinical agent performance on interactive reasoning over MIMIC-IV electronic health records. It consists of about 52K tasks across six clinical intents, requiring agents to execute SQL or Python in a sandboxed environment to answer patient- and population-level queries. Scoring is based on exact-match accuracy against expected outcomes.","whyItMatters":"Existing clinical benchmarks often rely on simplified, static SQL generation, failing to reflect real-world EHR complexity. EHR-Complex introduces interactive, multi-step reasoning tasks with compositional queries, revealing that state-of-the-art models achieve only 62.3% accuracy and exhibit fragility under repeated sampling, highlighting significant room for improvement in robust clinical reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6a7a4bcc87590e1d877d8fb6b3ed8b222b7c1bf3f1e614ed94e9d017e0713e3c"},"motivation":"Clinical agents promise to democratize access to electronic health records (EHRs), yet existing benchmarks fail to reflect the complexity of practical EHR analysis, e.g., often operating on idealized, clean EHRs via static SQL generation rather than interactive execution.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.23301","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_ehrbench_146544b7","familyId":"bmf_31bbb2b2bbf6","name":"EHRBench","oneLine":"EHRBench evaluates LLM-based clinical decision-making using nearly 1M QA items generated from real EHR trajectories, covering diagnosis, treatment, and prognosis tasks.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30637","pdf":"https://arxiv.org/pdf/2605.30637","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30637"},"evidence":{"snippet":"To fill the gaps, we introduce EHRBench, an automated and reliable EHR-grounded benchmark for evaluating LLM-based clinical decision-making at scale.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30637"},"ranking":{},"description":"EHRBench evaluates LLM-based clinical decision-making using nearly 1M QA items generated from real EHR trajectories, covering diagnosis, treatment, and prognosis tasks.","whyItMatters":"Existing clinical decision benchmarks often lack scale and reliability; EHRBench provides a large-scale, EHR-grounded evaluation that tests models on practical inference tasks, offering insights into model capabilities and gaps for clinical deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b5c1ad34c6617212ded22b5148ce02e459269f2c9de115937f1f9926d4336041"},"motivation":"Clinical decision-making (CDM) is central to real-world clinical workflows, where clinicians infer diagnoses, select treatments, or anticipate future health outcomes under incomplete evidence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30637","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_ehrnote-chatqa_748f904b","familyId":"bmf_6bcd9830c15b","name":"EHRNote-ChatQA","oneLine":"EHRNote-ChatQA is a benchmark for evidence-grounded multi-turn clinical QA over longitudinal discharge summaries. Built from MIMIC-IV, it includes 967 patient-level samples and 16,072 expert-verified QA pairs across eight clinical categories, with evidence-grounding QA pairs.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15735","pdf":"https://arxiv.org/pdf/2606.15735","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.15735"},"evidence":{"snippet":"We introduce EHRNote-ChatQA, the first benchmark for evidence-grounded multi-turn clinical question answering over patients' multiple discharge summaries.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15735"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EHRNote-ChatQA is a benchmark for evidence-grounded multi-turn clinical QA over longitudinal discharge summaries. Built from MIMIC-IV, it includes 967 patient-level samples and 16,072 expert-verified QA pairs across eight clinical categories, with evidence-grounding QA pairs.","whyItMatters":"EHRNote-ChatQA addresses the gap in evaluating clinical QA systems on multi-turn, evidence-grounded reasoning over multiple documents, which reflects real clinical review workflows. It provides a rigorous benchmark for assessing evidence grounding and the compounding of errors over turns.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"30890303f5a28845d1ffb75f3a6909cb92ebca9063463fb54f4261ca16723aea"},"motivation":"Discharge summaries are crucial clinical documents containing the context of a patient's overall hospital stay, and are routinely reviewed by medical experts for patient readmission, ongoing care, and diagnostic decision-making.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15735","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_eibench_cb3bf853","familyId":"bmf_1505489eac0d","name":"EIBench","oneLine":"EIBench is a simulator-based benchmark for interactive emotion management, containing 2,222 scenarios across a 2x2 taxonomy of Support, Defense, Repair, and Charm. It evaluates LLM agents on multi-turn dialogue where a user simulator updates emotion-relation states and provides anchor-based scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15532","pdf":"https://arxiv.org/pdf/2606.15532","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.15532"},"evidence":{"snippet":"We introduce EIBench, a simulator-based benchmark for interactive emotion management.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15532"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"EIBench is a simulator-based benchmark for interactive emotion management, containing 2,222 scenarios across a 2x2 taxonomy of Support, Defense, Repair, and Charm. It evaluates LLM agents on multi-turn dialogue where a user simulator updates emotion-relation states and provides anchor-based scoring.","whyItMatters":"EIBench fills the gap in evaluating emotional intelligence beyond static understanding, focusing on interactive emotion management over multiple turns. Its simulator provides both outcome and dense turn-level feedback, enabling training and evaluation in a unified environment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f83dd1f01dbe8123af75524a9e92e5e33cf36cd022a2ea59799fa35048fcb238"},"motivation":"Emotional intelligence (EI) in Large Language Models (LLMs) is often evaluated through static understanding tasks or single-response dialogue generation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15532","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_elbench_6c33e0fa","familyId":"bmf_3e9cf94fe4c5","name":"ELBench","oneLine":"ELBench is a benchmark for education-facing large language models, evaluating General Capability, Safety and Trustworthiness, Basic Education, and High-Level Cultivation under a common protocol with 2,939 items.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Inspectable","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09548","pdf":"https://arxiv.org/pdf/2608.09548","project":null,"code":null,"data":"https://huggingface.co/datasets/ZeroLoss-Lab/ELBench","hfPaper":"https://huggingface.co/papers/2608.09548"},"evidence":{"snippet":"We introduce ELBench, the first benchmark to evaluate all four requirements (General Capability, Safety and Trustworthiness, Basic Education, and High-Level Cultivation) on the same models under a common protocol, combining curated public sources with newly synthesized safety and cultivation data.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":64,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2608.09548"},"ranking":{"30d":{"score":49,"rank":33,"coverage":0.15,"confidence":"Low","datasetDownloadRank":25,"datasetRankPopulation":30},"90d":{"score":45,"rank":106,"coverage":0.3,"confidence":"Low","datasetDownloadRank":54,"datasetRankPopulation":66}},"description":"ELBench is a benchmark for education-facing large language models, evaluating General Capability, Safety and Trustworthiness, Basic Education, and High-Level Cultivation under a common protocol with 2,939 items.","whyItMatters":"It fills the gap of integrated evaluation for education-facing models, which require accuracy, safety, instructional usefulness, and pedagogical alignment simultaneously.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"efb65a84f9e941745f3e6784878ce52e4a604593d16968449b1d92c1dde37136"},"motivation":"Large language models are increasingly deployed in education as tutors, teaching assistants, and content generators.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09548","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ZeroLoss-Lab","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/datasets/ZeroLoss-Lab/ELBench","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_elephantbench_79ce069e","familyId":"bmf_80aadb10adc3","name":"ElephantBench","oneLine":"Closed-book knowledge probe with 1,094 questions evaluating recall of multiple divergent accounts for long-tail facts, with fixed C/P/F/K scoring metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.28478","pdf":"https://arxiv.org/pdf/2608.28478","project":null,"code":"https://github.com/Tencent/ElephantBench","data":null,"hfPaper":null},"evidence":{"snippet":"To address this gap, we introduce ElephantBench, a closed-book knowledge probe comprising 1,094 questions generated through an auditable graph-based pipeline.","reasonCodes":["exact named benchmark artifact released in abstract","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":11,"hfDailySubmittedAt":"2026-08-31T00:00:00.000Z","githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.28478"},"ranking":{"30d":{"score":30,"rank":82,"coverage":0.85,"confidence":"High"},"90d":{"score":30,"rank":244,"coverage":0.7,"confidence":"Medium"}},"description":"Closed-book knowledge probe with 1,094 questions evaluating recall of multiple divergent accounts for long-tail facts, with fixed C/P/F/K scoring metrics.","whyItMatters":"Provides a reproducible probe for diagnosing epistemic myopia in language models, with code and data available for direct use.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T04:51:30.440525Z","inputHash":"3e4161519e7a8e41bc2ceb02896b2a78698175e0edb8189fc83635a34512ecbf"},"motivation":"Factual question answering (QA) typically assumes a single canonical answer, obscuring whether large language models (LLMs) retain divergent accounts of long-tail facts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T04:51:30.440525Z","model":"deepseek-v4-pro","decisionReason":"GitHub repository and Hugging Face dataset provide complete data, code, and evaluation commands for fixed-scoring reuse.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce ElephantBench, a closed-book knowledge probe comprising 1,094 questions generated through an auditable graph-based pipeline."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.28478","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T01:03:30.163531Z"},"attentionForecast":{"score":70,"confidence":"Medium","horizon":"7d","reason":"Unique focus on long-tail divergent knowledge with auditable construction and open code, appealing to factuality researchers."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e251a7a05ee81b84","familyId":"catalog_family_e251a7a05ee81b84","name":"EMB","oneLine":"Evaluating agents on Excel-based financial modeling tasks","description":"Evaluating agents on Excel-based financial modeling tasks","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/emb","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e251a7a05ee81b84"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/emb"}],"catalogSources":[{"catalog":"benchlm","sourceId":"emb","url":"https://benchlm.ai/benchmarks/emb","paperUrl":"https://www.vals.ai/benchmarks/emb","year":"2026","fullName":"Vals EMB","format":"Accuracy score","tasks":"Excel-based financial modeling tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_embodied3dbench_d183e439","familyId":"bmf_1a1023e37b98","name":"Embodied3DBench","oneLine":"Embodied3DBench is a robot-centric benchmark for low-level spatial intelligence in embodied 3D environments. It includes 6 task categories: Grounding, Spatial Relation Prediction, Multi-view Correspondence, Affordance Prediction, Grasp Point Prediction, and Trajectory Prediction, with 21k QA pairs across 12 subcategories.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Geometric reasoning"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29074","pdf":"https://arxiv.org/pdf/2605.29074","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29074"},"evidence":{"snippet":"We introduce Embodied3DBench, a robot-centric benchmark targeting low-level spatial intelligence in embodied 3D environments.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29074"},"ranking":{},"description":"Embodied3DBench is a robot-centric benchmark for low-level spatial intelligence in embodied 3D environments. It includes 6 task categories: Grounding, Spatial Relation Prediction, Multi-view Correspondence, Affordance Prediction, Grasp Point Prediction, and Trajectory Prediction, with 21k QA pairs across 12 subcategories.","whyItMatters":"Current VLMs show strong high-level spatial reasoning but lack interaction-oriented perception. Embodied3DBench reveals this gap and provides a scalable training dataset for improvement, enabling systematic evaluation and advancement of interaction-aware multimodal systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"49d23f33ffa8b41b6907b442b42771c8d6a65a7cab2d758b5f8fa1190032c601"},"motivation":"Are current Vision Language Models (VLMs) ready to comprehend and reason about complex embodied interactions in 3D environments?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29074","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_fd305ae2bc69d610","familyId":"catalog_family_fd305ae2bc69d610","name":"EmbSpatialBench","oneLine":"EmbSpatialBench evaluates embodied spatial understanding and reasoning capabilities.","description":"EmbSpatialBench evaluates embodied spatial understanding and reasoning capabilities.","area":"Mathematical Reasoning","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":[],"capabilities":[],"topics":["Spatial Reasoning","Embodied","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/embspatialbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fd305ae2bc69d610"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/embspatialbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"embspatialbench","url":"https://llm-stats.com/benchmarks/embspatialbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["spatial reasoning","embodied","vision"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"bm_emem-bench_594a2f57","familyId":"bmf_3800709e05be","name":"eMEM-Bench","oneLine":"eMEM-Bench v1 evaluates embodied memory systems through 988 probes across eight cognitive-psychology paradigms (DRM lures, pattern separation, pattern completion, source monitoring, context-dependent retrieval, long-horizon interference, serial position, foil retention curve) in ProcTHOR-10K scenes.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03374","pdf":"https://arxiv.org/pdf/2606.03374","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03374"},"evidence":{"snippet":"In addition we introduce eMEM-Bench v1, a benchmark we construct over ProcTHOR-10K scenes for embodied memory evaluation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03374"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"eMEM-Bench v1 evaluates embodied memory systems through 988 probes across eight cognitive-psychology paradigms (DRM lures, pattern separation, pattern completion, source monitoring, context-dependent retrieval, long-horizon interference, serial position, foil retention curve) in ProcTHOR-10K scenes.","whyItMatters":"Existing agent memory benchmarks lack diagnostic depth. eMEM-Bench provides interpretable results tied to memory-systems literature, allowing fine-grained comparison of embodied memory architectures beyond surface-task accuracy.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"80342a0435f2305b58474c90aaa881bd9bd39666b1bacf0636861f944d0e4c84"},"motivation":"We present eMEM (Embodied Memory), a hybrid graph-based memory system for embodied agents operating in physical environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03374","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_f4aa0655cdb8d4fc","familyId":"catalog_family_f4aa0655cdb8d4fc","name":"EMMA","oneLine":"EMMA (Enhanced MultiModal reAsoning) is a benchmark for organic multimodal reasoning across mathematics, physics, chemistry, and coding.","description":"EMMA (Enhanced MultiModal reAsoning) is a benchmark for organic multimodal reasoning across mathematics, physics, chemistry, and coding.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/emma","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f4aa0655cdb8d4fc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/emma"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"emma","url":"https://llm-stats.com/benchmarks/emma","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","multimodal","reasoning","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_empath_eb9c1b58","familyId":"bmf_85697bf1d578","name":"EMPATH","oneLine":"EMPATH evaluates safety of emotional-support chatbots via auditor-generated multi-turn conversations scored on 19 metrics across crisis handling, therapeutic quality, conversational integrity, emotional safety, and cultural adaptation.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.30256","pdf":"https://arxiv.org/pdf/2606.30256","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.30256"},"evidence":{"snippet":"We present EMPATH, a benchmark for safety evaluation of emotional-support chatbots.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.30256"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EMPATH evaluates safety of emotional-support chatbots via auditor-generated multi-turn conversations scored on 19 metrics across crisis handling, therapeutic quality, conversational integrity, emotional safety, and cultural adaptation.","whyItMatters":"Safety evaluation for emotional-support chatbots needs multilingual, multi-turn metrics; EMPATH appears to address that gap.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8c0e01b95d534c5c4990283bc510140cad34ead7a421e06e73b6d5c6b673f5bd"},"motivation":"Safety benchmarks often buy scalability by fixing the prompt, the language, and the turn structure.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.30256","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_on-the-robustness-of-temporal-vision-langu_c61c2c1d","familyId":"bmf_51771cb8904f","name":"Endo-C6","oneLine":"Evaluates temporal vision-language models on surgical endoscopy video understanding under six realistic corruptions, using public videos and standardized prompts.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Robustness"],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":0.45,"links":{"report":"https://arxiv.org/abs/2608.14262","pdf":"https://arxiv.org/pdf/2608.14262","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce Endo-C6, a compact corruption benchmark of six endoscopy-realistic perturbations evaluated at a fixed high severity, and apply it to public Gastrointestinal (GI) endoscopy and laparoscopic cholecystectomy videos.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14262"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates temporal vision-language models on surgical endoscopy video understanding under six realistic corruptions, using public videos and standardized prompts.","whyItMatters":"Provides a standardized robustness benchmark for clinical vision-language systems, exposing worst-case performance degradation under clinically relevant artifacts.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"88732cbde55c84ce1b15d56636c10b7c1bfc65914c6e85bc4f329e5454673d39"},"motivation":"Temporal vision-language models (TVLMs) offer a reusable, prompt-based interface for surgical video understanding, yet, their robustness under clinically realistic acquisition artifacts in endoscopy remains insufficiently characterized.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is described as reproducible with public videos and standardized protocols, and the paper was accepted to MICCAI 2026, indicating a formal evaluation setup.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce Endo-C6, a compact corruption benchmark of six endoscopy-realistic perturbations"},"publication":{"status":"acceptance_claimed","venue":"MICCAI 2026","evidence":"Accepted to MICCAI 2026","evidenceUrl":"https://arxiv.org/abs/2608.14262","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-19T11:05:13.395391Z"},"venueAttempts":[{"venueName":"MICCAI 2026","reviewStatus":"accepted","decisionRaw":"Accepted to MICCAI 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.14262","observedAt":"2026-08-19T11:05:13.395391Z","rawValue":"Accepted to MICCAI 2026","level":"author-claim"}]}],"attentionForecast":{"score":45,"confidence":"Medium","horizon":"7d","reason":"Acceptance at MICCAI and the clinical relevance of robustness evaluation are likely to generate moderate interest, though the benchmark is specialized."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_enpmr-bench_0b8d24ff","familyId":"bmf_d98b3aef5c80","name":"ENPMR-Bench","oneLine":"Benchmark for emotional need-aware proactive memory retrieval in support agents, with 1,800+ dialogues mapped to Maslow's hierarchy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.27240","pdf":"https://arxiv.org/pdf/2605.27240","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27240"},"evidence":{"snippet":"In this work, we introduce ENPMR-Bench, a benchmark for evaluating Emotional Need-aware Proactive Memory Retrieval (ENPMR), a core capability that enables agents to infer users' latent emotional needs and proactively retrieve appropriate memories to support empathetic interaction.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27240"},"ranking":{},"description":"Benchmark for emotional need-aware proactive memory retrieval in support agents, with 1,800+ dialogues mapped to Maslow's hierarchy.","whyItMatters":"Current memory retrieval is factual, neglecting emotional needs. ENPMR-Bench evaluates proactive retrieval for empathetic interaction.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"73eb640026de31f2ab73c106d06cca05d1bc96004993c4713310269cac7240bb"},"motivation":"Memory-augmented language agents are increasingly deployed in affective applications such as emotional support, where understanding and responding to users' latent emotional needs is critical.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27240","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_enterpriseclawbench_252ea1ca","familyId":"bmf_7de792287b02","name":"EnterpriseClawBench","oneLine":"Evaluates coding agents on realistic enterprise workflows using 852 reproducible tasks with recovered fixtures, prompts, role classes, skill subclasses, hard rules, and semantic rubrics. Scoring covers artifact delivery, visual quality, cost, runtime, and skill transfer.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.23654","pdf":"https://arxiv.org/pdf/2606.23654","project":null,"code":"https://github.com/FrontisAI/EnterpriseClawBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.23654"},"evidence":{"snippet":"We introduce EnterpriseClawBench, an enterprise agent benchmark constructed from proprietary, real-world agent sessions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":80,"hfDailySubmittedAt":"2026-06-23T00:00:00.000Z","githubStars":47,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.23654"},"ranking":{"90d":{"score":56,"rank":33,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates coding agents on realistic enterprise workflows using 852 reproducible tasks with recovered fixtures, prompts, role classes, skill subclasses, hard rules, and semantic rubrics. Scoring covers artifact delivery, visual quality, cost, runtime, and skill transfer.","whyItMatters":"Enterprise agent evaluation often collapses performance into a single score, ignoring cost, runtime, and artifact quality. This benchmark provides a multi-faceted scoring contract and a reusable construction/evaluation protocol, enabling practical comparisons of harness-model systems in real workplace settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2da25124a1691b0d43a4edc485fe703246c2b9b27ab36a755cf69301364cde2b"},"motivation":"Enterprise agents increasingly operate inside workspaces: they read heterogeneous files, invoke tools, and deliver business artifacts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.23654","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FrontisAI","organizationType":"company-research-lab","sourceUrl":"https://github.com/FrontisAI/EnterpriseClawBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_enterprisemem-bench_96086215","familyId":"bmf_78a3d7491203","name":"EnterpriseMem-Bench","oneLine":"EnterpriseMem-Bench is a multi-turn Text-to-SQL benchmark with 300 sessions and 1,400 turns across three enterprise domains, featuring deterministic ground truth and per-turn memory-critical annotations.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26394","pdf":"https://arxiv.org/pdf/2605.26394","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.26394"},"evidence":{"snippet":"We introduce EnterpriseMem-Bench, a multi-turn Text-to-SQL benchmark of 300 sessions and 1,400 turns built programmatically from three enterprise domains (BIRD financial, SEC EDGAR, Northwind), with deterministic ground truth and per-turn memory-critical annotation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26394"},"ranking":{},"description":"EnterpriseMem-Bench is a multi-turn Text-to-SQL benchmark with 300 sessions and 1,400 turns across three enterprise domains, featuring deterministic ground truth and per-turn memory-critical annotations.","whyItMatters":"It addresses the lack of multi-turn evaluation in Text-to-SQL, providing insights into memory architecture effects and model performance degradation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dafbef18e13b541cae79dce70f6abdeb427944fc926fdef83e553e75f706e941"},"motivation":"Multi-turn Text-to-SQL is central to enterprise analytics yet remains predominantly evaluated in single-turn settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26394","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_enterpriserag_49e3b609","familyId":"bmf_11382aad9957","name":"EnterpriseRAG","oneLine":"The benchmark evaluates LLM instruction adherence and robustness in enterprise retrieval scenarios, using 983 expert-validated samples across six domains, simulating retrieval noise, knowledge gaps, and factual conflicts.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval","Robustness","Factuality"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.11584","pdf":"https://arxiv.org/pdf/2608.11584","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.11584"},"evidence":{"snippet":"We introduce EnterpriseRAG, a benchmark of 983 expert-validated samples across six domains that systematically simulates three failure modes absent from prior work: retrieval noise, knowledge gaps, and factual conflicts, coupled with complex instructions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11584"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"The benchmark evaluates LLM instruction adherence and robustness in enterprise retrieval scenarios, using 983 expert-validated samples across six domains, simulating retrieval noise, knowledge gaps, and factual conflicts.","whyItMatters":"Existing RAG benchmarks assume clean retrieval and simple queries, failing to capture production conditions. This benchmark addresses the gap by measuring holistic compliance under non-ideal conditions, informing deployment decisions for enterprise-scale RAG systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0d95abb771fcce5a6b9fac9c7f67ebbd1c0e79f449ded04813552f41eefac32a"},"motivation":"Enterprise RAG deployments face a critical reliability gap: while LLMs satisfy 80% of individual constraints, only 26.8% of responses meet all requirements simultaneously, revealing a 57-point orchestration gap.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11584","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness","Search & Retrieval"],"domainScope":"general"},{"id":"bm_entlore-a-graph-grounded-benchmark-for-lat_335f471f","familyId":"bmf_ce281d2f4306","name":"ENTLORE","oneLine":"Evaluates enterprise question answering over 2,341 documents and 907 questions spanning explicit lookup, cross-source composition, and latent organizational reasoning.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.10679","pdf":"https://arxiv.org/pdf/2608.10679","project":null,"code":"https://github.com/scitix/entlore","data":null,"hfPaper":null},"evidence":{"snippet":"We introduce ENTLORE, a graph-grounded benchmark construction framework that reconstructs an audited enterprise world from routine documents, authoritative organizational tables, and operational records.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","public artifact URL","named entity is a model, method, framework, or dataset used in evaluation"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":6,"hfDailySubmittedAt":"2026-08-18T00:00:00.000Z","githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10679"},"ranking":{"30d":{"score":29,"rank":83,"coverage":0.85,"confidence":"High"},"90d":{"score":31,"rank":238,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates enterprise question answering over 2,341 documents and 907 questions spanning explicit lookup, cross-source composition, and latent organizational reasoning.","whyItMatters":"Tests whether models can recover implicit organizational relations from documents, filling a gap left by benchmarks that only compose stated facts.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"294f432e44a040d5a0e0e4d502422b82336f08e1aca6dd70b229836732fead44"},"motivation":"Enterprise question answering is framed as retrieving internal documents and generating grounded answers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"Benchmark with released dataset, code, and reproducible evaluation protocol on GitHub and Hugging Face, with complete golden answers and proof certificates.","canonicalNameSource":"official_readme","canonicalNameEvidence":"EntLORE A Graph-Grounded Benchmark for Latent Organizational Reasoning in Enterprise Question Answering"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10679","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"Novel enterprise QA benchmark with strong artifacts and a clear latent reasoning contribution, though niche enterprise focus may limit breadth."},"evaluationMode":"public_reusable","publishers":[{"name":"ScitiX.ai","organizationType":"company-research-lab","sourceUrl":"https://github.com/scitix/entlore","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_entsql_088384b0","familyId":"bmf_c6be492ed41d","name":"EntSQL","oneLine":"EntSQL evaluates text-to-SQL systems on enterprise knowledge grounding, with 1,066 Chinese-English examples across five business domains requiring private business knowledge. Systems generate SQL from questions and schema, with some provided long-form documents.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03363","pdf":"https://arxiv.org/pdf/2606.03363","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03363"},"evidence":{"snippet":"We introduce EntSQL, an enterprise-oriented Text-to-SQL benchmark for evaluating long-context grounding over proprietary business documents.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03363"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EntSQL evaluates text-to-SQL systems on enterprise knowledge grounding, with 1,066 Chinese-English examples across five business domains requiring private business knowledge. Systems generate SQL from questions and schema, with some provided long-form documents.","whyItMatters":"Existing text-to-SQL benchmarks overlook enterprise scenarios where SQL generation depends on proprietary business knowledge. EntSQL measures the ability to ground SQL generation in long-context enterprise documents, revealing a significant performance gap in current systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"18604bf22b7018a140fdaf137b507c36ec00c009dd6335570dac90be6dc09a21"},"motivation":"Text-to-SQL enables natural language access to databases, and recent LLMs have substantially advanced its capabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03363","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Long Context & Memory"],"domainScope":"general"},{"id":"bm_envship_5107b438","familyId":"bmf_467188c4c8c6","name":"EnvShip","oneLine":"EnvShip is a unified multi-region framework for context-aware and cross-region vessel trajectory forecasting, with standardized tracks and evaluation protocols using public AIS data.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-13","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.15240","pdf":"https://arxiv.org/pdf/2606.15240","project":null,"code":null,"data":"https://huggingface.co/datasets/mark000071/envship_v2_datasets","hfPaper":"https://huggingface.co/papers/2606.15240"},"evidence":{"snippet":"EnvShip provides a common and reproducible testbed for vessel trajectory forecasting.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":317,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2606.15240"},"ranking":{"90d":{"score":49,"rank":68,"coverage":0.3,"confidence":"Low","datasetDownloadRank":29,"datasetRankPopulation":66}},"description":"EnvShip is a unified multi-region framework for context-aware and cross-region vessel trajectory forecasting, with standardized tracks and evaluation protocols using public AIS data.","whyItMatters":"Existing studies use incompatible preprocessing and evaluation settings, making results difficult to compare. EnvShip provides a common, reproducible testbed for vessel trajectory forecasting, enabling fair comparison across methods and regions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2eb2534910e8fc56505bdc8b26b89179a1df9f475981ff6e54c6a55d50403a10"},"motivation":"Accurate vessel trajectory forecasting is essential for maritime situational awareness, navigation safety, traffic management, and autonomous navigation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15240","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_epibench_00d62089","familyId":"bmf_12b5a09e9b7d","name":"EpiBench","oneLine":"EpiBench evaluates LLMs' epitope reasoning from antibody and antigen sequences across five tasks: targetable region discovery, epitope identification, binning, functional assessment, and escape assessment.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.06022","pdf":"https://arxiv.org/pdf/2608.06022","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.06022"},"evidence":{"snippet":"To address this gap, we introduce EpiBench, a closed-book, sequence-based, and automatically scorable benchmark for evaluating epitope reasoning in LLMs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06022"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EpiBench evaluates LLMs' epitope reasoning from antibody and antigen sequences across five tasks: targetable region discovery, epitope identification, binning, functional assessment, and escape assessment.","whyItMatters":"Previous epitope resources focus on isolated prediction tasks and do not evaluate epitope-centered decisions across the antibody development workflow. EpiBench provides a closed-book, automatically scorable benchmark for this purpose.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b85c65ee3ef651e110795f44fe1baa689b0c011473ea549ca08f2448835915c3"},"motivation":"Epitopes determine where antibodies bind antigens and shape downstream therapeutic properties such as functional blockade and escape resistance, making epitope understanding central to antibody drug discovery.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06022","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_epibench_be78266a","familyId":"bmf_12b5a09e9b7d","name":"EpiBench","oneLine":"EpiBench evaluates AI agents on short-horizon epigenomics analysis tasks across CUT&Tag/CUT&RUN, ATAC-seq, ChIP-seq, and DNA methylation workflows, with deterministically gradable answers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13602","pdf":"https://arxiv.org/pdf/2606.13602","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.13602"},"evidence":{"snippet":"We introduce EpiBench, a verifiable benchmark for short-horizon epigenomics analysis.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13602"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EpiBench evaluates AI agents on short-horizon epigenomics analysis tasks across CUT&Tag/CUT&RUN, ATAC-seq, ChIP-seq, and DNA methylation workflows, with deterministically gradable answers.","whyItMatters":"It addresses the lack of verifiable benchmarks for epigenomics analysis agents, providing a basis for comparing agent performance on complex scientific tasks that require domain-specific judgment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"034189a803fb4170a9fc7ed058dbf2356b41f6195ef3fa0f3290efd8c59bea8b"},"motivation":"We introduce EpiBench, a verifiable benchmark for short-horizon epigenomics analysis.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13602","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_ab06169b3c2aeee2","familyId":"catalog_family_ab06169b3c2aeee2","name":"EQ-Bench","oneLine":"EQ-Bench is an LLM-judged test evaluating active emotional intelligence abilities, understanding, insight, empathy, and interpersonal skills. The test set contains 45 challenging roleplay scenarios, most of which constitute pre-written prompts spanning 3 turns. The benchmark evaluates the performance of models by validating responses against several criteria and conducts pairwise comparisons to report a normalized Elo computation for each model.","description":"EQ-Bench is an LLM-judged test evaluating active emotional intelligence abilities, understanding, insight, empathy, and interpersonal skills. The test set contains 45 challenging roleplay scenarios, most of which constitute pre-written prompts spanning 3 turns. The benchmark evaluates the performance of models by validating responses against several criteria and conducts pairwise comparisons to report a normalized Elo computation for each model.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Roleplay","General","Creativity","Writing"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/eq-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ab06169b3c2aeee2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/eq-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"eq-bench","url":"https://llm-stats.com/benchmarks/eq-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","roleplay","general","creativity","writing"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_bdfc2c14b5fe0abc","familyId":"catalog_family_bdfc2c14b5fe0abc","name":"EQ-Bench 4","oneLine":"A multi-turn benchmark of emotional and social intelligence using synthetic personas and pairwise LLM judging.","description":"A multi-turn benchmark of emotional and social intelligence using synthetic personas and pairwise LLM judging.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://eqbench.com/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bdfc2c14b5fe0abc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/eqbench4"}],"catalogSources":[{"catalog":"benchlm","sourceId":"eqBench4","url":"https://benchlm.ai/benchmarks/eqbench4","paperUrl":"https://eqbench.com/","year":"2026","fullName":"EQ-Bench 4","format":"Pairwise Elo","tasks":"120 multi-turn persona scenarios","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_erbench_e22588c9","familyId":"bmf_58b25273f95f","name":"ERBench","oneLine":"ERBench is an evaluation framework for equation discovery algorithms, assessing recovery of groundtruth formulas under varying dimensionality, sampling size, distribution, and domain.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09276","pdf":"https://arxiv.org/pdf/2606.09276","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.09276"},"evidence":{"snippet":"To fill this gap, we introduce the Equation Recovery Benchmark (ERBench), a new evaluation framework designed to rigorously assess algorithms explicitly targeting the task of equation discovery.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09276"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"ERBench is an evaluation framework for equation discovery algorithms, assessing recovery of groundtruth formulas under varying dimensionality, sampling size, distribution, and domain.","whyItMatters":"Existing symbolic regression benchmarks have few public groundtruth formulas and limited robustness testing; ERBench aims to provide a more rigorous evaluation for algorithm practitioners.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"68d4418438db88d0dc070d2f8e2582d8e611fbc1c41ee78630a211bf222c206e"},"motivation":"Equation discovery aims to automate the discovery of scientific models in the form of mathematical equations from data.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09276","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ergeobench_37de7936","familyId":"bmf_323d61775db6","name":"ERGeoBench","oneLine":"ERGeoBench evaluates vision-driven embodied geo-localization in MLLMs with 2,207 globally distributed street-view panoramas under single-view, panorama-view, and embodied-view settings. It measures foundational perception, spatial awareness, common sense reasoning, and geo-localization reasoning.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.31251","pdf":"https://arxiv.org/pdf/2605.31251","project":"https://kaixuewen.github.io/ERGeoBench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.31251"},"evidence":{"snippet":"We introduce ERGeoBench, a diagnostic benchmark for vision-driven embodied geo-localization.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.31251"},"ranking":{},"description":"ERGeoBench evaluates vision-driven embodied geo-localization in MLLMs with 2,207 globally distributed street-view panoramas under single-view, panorama-view, and embodied-view settings. It measures foundational perception, spatial awareness, common sense reasoning, and geo-localization reasoning.","whyItMatters":"Embodied geo-localization is underexplored due to lack of fine-grained evaluation. ERGeoBench provides a unified diagnostic framework that reveals current MLLMs struggle with fine-grained perceptual operations and metric localization, supporting progress in integrated perception and spatial reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"19f60e4b9df0eb40d3e7709010980223210ff749e827affe8ff9644b57c89eec"},"motivation":"Multimodal large language models (MLLMs) have shown strong potential as embodied agents, yet embodied geo-localization remains underexplored due to the lack of fine-grained evaluation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.31251","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ERGeoBench Team","organizationType":"academic-lab","sourceUrl":"https://kaixuewen.github.io/ERGeoBench/","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_a5789abbeab1284a","familyId":"catalog_family_a5789abbeab1284a","name":"ERQA","oneLine":"Embodied Reasoning Question Answering benchmark consisting of 400 multiple-choice visual questions across spatial reasoning, trajectory reasoning, action reasoning, state estimation, and multi-view reasoning for evaluating AI capabilities in physical world interactions","description":"Embodied Reasoning Question Answering benchmark consisting of 400 multiple-choice visual questions across spatial reasoning, trajectory reasoning, action reasoning, state estimation, and multi-view reasoning for evaluating AI capabilities in physical world interactions","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Reasoning","Spatial Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a5789abbeab1284a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/erqa"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/erqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"erqa","url":"https://benchlm.ai/benchmarks/erqa","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"ERQA","format":"Grounded image reasoning","tasks":"Evidence-based visual QA","successorKey":null},{"catalog":"llm-stats","sourceId":"erqa","url":"https://llm-stats.com/benchmarks/erqa","datasetSlug":"erqa","versionCount":1,"subsetCount":1,"rowCount":null,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":["multimodalGrounded","reasoning","spatial reasoning","vision"],"catalogModelCount":25,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_erqa-plus_148e4067","familyId":"bmf_43898951b3ad","name":"ERQA-Plus","oneLine":"ERQA-Plus evaluates embodied reasoning in AI systems with 1,766 question-answer instances grounded in 711 robot-centric images, organized by a taxonomy covering perceptual, action-centric, social-interaction, navigation-environmental, and commonsense reasoning. Scoring uses overall accuracy and SBERT similarity.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.17639","pdf":"https://arxiv.org/pdf/2606.17639","project":null,"code":"https://github.com/LUNAProject22/erqa-plus","data":"https://huggingface.co/datasets/huggingdas/erqa-plus","hfPaper":"https://huggingface.co/papers/2606.17639"},"evidence":{"snippet":"We present ERQA-Plus, a diagnostic benchmark for reasoning in embodied AI.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":1488,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2606.17639"},"ranking":{"90d":{"score":53,"rank":45,"coverage":0.3,"confidence":"Low","datasetDownloadRank":14,"datasetRankPopulation":66}},"description":"ERQA-Plus evaluates embodied reasoning in AI systems with 1,766 question-answer instances grounded in 711 robot-centric images, organized by a taxonomy covering perceptual, action-centric, social-interaction, navigation-environmental, and commonsense reasoning. Scoring uses overall accuracy and SBERT similarity.","whyItMatters":"Existing visual and embodied QA benchmarks often lack control over reasoning dependencies, making it hard to distinguish genuine embodied reasoning from shortcut-driven pattern matching. ERQA-Plus provides a fine-grained diagnostic to identify strengths and weaknesses in specific reasoning categories, informing model development and deployment decisions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7791fd2963a0f20915e4425ee5ea4b82d7d2e57de5e196da5494206fd71fdd8e"},"motivation":"Generalist embodied agents require more than object recognition: they must reason about spatial relations, actions, procedures, human intentions, environmental constraints, and commonsense consequences from situated visual observations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.17639","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LUNAProject22","organizationType":"community","sourceUrl":"https://github.com/LUNAProject22/erqa-plus","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_erunderstand_657f59fd","familyId":"bmf_961dfc017065","name":"ERUnderstand","oneLine":"ERUnderstand evaluates vision-language models on structured understanding of entity-relationship diagrams. The benchmark contains 2,960 diagrams across curated educational sources, real-world schemas, and synthetic data, with standardized machine-readable JSON annotations. Scoring uses F1, BLEU, and graph edit distance to measure how accurately models recover schema elements such as entities, relationships, attributes, and extended ER constructs.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.24707","pdf":"https://arxiv.org/pdf/2607.24707","project":null,"code":"https://github.com/salinaria/ERUnderstand","data":null,"hfPaper":"https://huggingface.co/papers/2607.24707"},"evidence":{"snippet":"We introduce ERUnderstand, the first large-scale benchmark for structured understanding of ER diagrams, comprising 2,960 diagrams collected from curated educational sources, real-world schemas, and synthetically generated examples spanning diverse domains, notations, complexity levels, and Extended Entity-Relationship (EER) constructs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24707"},"ranking":{"90d":{"score":30,"rank":242,"coverage":0.7,"confidence":"Medium"}},"description":"ERUnderstand evaluates vision-language models on structured understanding of entity-relationship diagrams. The benchmark contains 2,960 diagrams across curated educational sources, real-world schemas, and synthetic data, with standardized machine-readable JSON annotations. Scoring uses F1, BLEU, and graph edit distance to measure how accurately models recover schema elements such as entities, relationships, attributes, and extended ER constructs.","whyItMatters":"ER diagrams are central to database design but their image-based nature impedes automated processing. Existing VLM benchmarks do not focus on structured schema extraction from ERDs, and no public benchmark provides a standardized protocol for this task. ERUnderstand fills that gap with a reusable dataset and evaluation toolkit, enabling reproducible comparison of VLM performance on conceptual database schemas.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"90f18217e551a6266cad1ded9eeda84755e1af9dbf872ef664ec3fb6683db4e5"},"motivation":"Entity-Relationship Diagrams (ERDs) are central to conceptual database design, yet they are typically available only as rendered images rather than machine-readable schemas, limiting AI-assisted database engineering.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24707","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_escucha_bdd66fa8","familyId":"bmf_07214793264e","name":"ESCUCHA","oneLine":"ESCUCHA is a Spanish speech understanding benchmark evaluating LALMs across heterogeneous acoustic conditions and reasoning abilities. It includes 1,000 human-curated audio-question pairs spanning perceptual and reasoning categories, with multiple accents and non-normative speech.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.17812","pdf":"https://arxiv.org/pdf/2607.17812","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.17812"},"evidence":{"snippet":"We introduce ESCUCHA, the first Spanish speech understanding benchmark designed to evaluate LALMs across heterogeneous acoustic conditions and reasoning abilities.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.17812"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ESCUCHA is a Spanish speech understanding benchmark evaluating LALMs across heterogeneous acoustic conditions and reasoning abilities. It includes 1,000 human-curated audio-question pairs spanning perceptual and reasoning categories, with multiple accents and non-normative speech.","whyItMatters":"The benchmark addresses the lack of robust evaluation for Spanish speech understanding under realistic conditions. It provides practical value in assessing model performance across diverse acoustic environments and reasoning tasks, highlighting gaps relative to human performance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"58de95993e993265778207d70af3e7fafb275cd603138bd1e7434fb757a5617e"},"motivation":"As large audio language models (LALMs) advance, robust evaluation frameworks have become essential.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.17812","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_esf-bench_d9226708","familyId":"bmf_db75a85566d3","name":"ESF-Bench","oneLine":"ESF-Bench evaluates slot filling in enterprise contexts, covering 810 multi-turn dialogues and 6,530 slots across 8 domains, with a taxonomy of 57 challenging scenarios.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.23326","pdf":"https://arxiv.org/pdf/2607.23326","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.23326"},"evidence":{"snippet":"In this work, we introduce ESF-Bench, a challenging Enterprise Slot Filling benchmark consisting of 810 multi-turn samples and 6530 slots over 8 unique domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23326"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ESF-Bench evaluates slot filling in enterprise contexts, covering 810 multi-turn dialogues and 6,530 slots across 8 domains, with a taxonomy of 57 challenging scenarios.","whyItMatters":"Addresses the gap in evaluating LLMs for slot filling under real-world enterprise constraints and unexpected user behaviors, providing a standard for model comparison in this practical task.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2eb3e1b64b89447e78bf8bb9e9077712ec25218bcb3701efe1ea74d834ea5c3a"},"motivation":"The rapid rise of large language models (LLMs) has driven transformative adoption across enterprises.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.23326","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_esrp-bench_4847c977","familyId":"bmf_6edc05e43dce","name":"ESRP-Bench","oneLine":"This paper introduces Embodied Scene Rearrangement Planning (ESRP), a novel task requiring embodied agents to rearrange furniture in 3D scenes to match a target configuration using only egocentric observations and a top-down target layout.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Planning"],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.27371","pdf":"https://arxiv.org/pdf/2608.27371","project":"https://pie-lab.cn/ESRP/","code":"https://github.com/BIT-PIE/ESRP","data":"https://huggingface.co/datasets/serendipity800/ESRP-PD","hfPaper":"https://huggingface.co/papers/2608.27371"},"evidence":{"snippet":"To facilitate research, we present ESRP-Bench, a comprehensive benchmark built on OmniGibson featuring over 5,400 scene pairs and 8,200 objects.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":58,"hfDatasetLikes":1},"source":{"type":"arxiv","id":"2608.27371"},"ranking":{"30d":{"score":31,"rank":81,"coverage":0.7,"confidence":"Medium","datasetDownloadRank":29,"datasetRankPopulation":30},"90d":{"score":26,"rank":294,"coverage":0.85,"confidence":"High","datasetDownloadRank":59,"datasetRankPopulation":66}},"motivation":"This paper introduces Embodied Scene Rearrangement Planning (ESRP), a novel task requiring embodied agents to rearrange furniture in 3D scenes to match a target configuration using only egocentric observations and a top-down target layout.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"acceptance_claimed","venue":"publication in IEEE Robotics and Automation Letters (RA-L), 2026","evidence":"Accepted for publication in IEEE Robotics and Automation Letters (RA-L), 2026. Project page: https://bit-pie.github.io/ESRP/ Code: https://github.com/BIT-PIE/ESRP Dataset: https://huggingface.co/datasets/serendipity800/ESRP-PD","evidenceUrl":"https://arxiv.org/abs/2608.27371","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"publication in IEEE Robotics and Automation Letters (RA-L), 2026","reviewStatus":"accepted","decisionRaw":"Accepted for publication in IEEE Robotics and Automation Letters (RA-L), 2026. Project page: https://bit-pie.github.io/ESRP/ Code: https://github.com/BIT-PIE/ESRP Dataset: https://huggingface.co/datasets/serendipity800/ESRP-PD","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.27371","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted for publication in IEEE Robotics and Automation Letters (RA-L), 2026. Project page: https://bit-pie.github.io/ESRP/ Code: https://github.com/BIT-PIE/ESRP Dataset: https://huggingface.co/datasets/serendipity800/ESRP-PD","level":"author-claim"}]}],"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_euroexec_53125f62","familyId":"bmf_2d0194cba00a","name":"EuroExec","oneLine":"EuroExec is an expert-authored benchmark of 413 open-ended European executive tasks evaluated by human experts. It uses a multi-attribute rubric and preference ranking to compute a Solve Rate.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.04549","pdf":"https://arxiv.org/pdf/2608.04549","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.04549"},"evidence":{"snippet":"We dedicate more than 4,000 human expert hours to evaluate a selection of six frontier LLMs on a member of this class of problems: EuroExec, our introduced human expert-based benchmark composed of 413 open-ended long-form European executive tasks authored by 47 vetted domain experts, each question drawn from experience in a real case.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04549"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EuroExec is an expert-authored benchmark of 413 open-ended European executive tasks evaluated by human experts. It uses a multi-attribute rubric and preference ranking to compute a Solve Rate.","whyItMatters":"It addresses evaluation of open-ended, complex tasks where subjective judgment is key, but the lack of public artifacts makes its reuse uncertain.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b52fc3bf1603b151ec3e788412ca59c9da373be792b76784774be06a631a88a9"},"motivation":"Frontier LLMs are increasingly put to use on open-ended complex questions, different in nature from the ones they are typically evaluated on.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04549","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_euskal-agent-bench_06279148","familyId":"bmf_f335d1b163a1","name":"euskal-agent-bench","oneLine":"Euskal-agent-bench is a project evaluating LLM agents in Basque, Spanish, and English across 30 tasks with direct prompting and tool-using agent architectures, measuring capability and agentic-stack degradation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/GorkaFS13/euskal-agent-bench","pdf":null,"project":"https://gorkafs13.github.io/euskal-agent-bench/","code":"https://github.com/GorkaFS13/euskal-agent-bench","data":null,"hfPaper":null},"evidence":{"snippet":"euskal-agent-bench Do LLM agents work in Basque?","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:gorkafs13/euskal-agent-bench"},"ranking":{"30d":{"score":28,"rank":91,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":261,"coverage":0.55,"confidence":"Low"}},"description":"Euskal-agent-bench is a project evaluating LLM agents in Basque, Spanish, and English across 30 tasks with direct prompting and tool-using agent architectures, measuring capability and agentic-stack degradation.","whyItMatters":"It investigates whether agentic stacks (tool calling, output contracts) degrade in low-resource languages, relevant for production LLM systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-25T16:23:14.196185Z","inputHash":"0d30be5ce7bff8f872821e9dd5f86c138fcc8cebdbc8b228af0bb16a385cd8b5"},"motivation":"euskal-agent-bench Do LLM agents work in Basque?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T15:38:21.596820Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/GorkaFS13/euskal-agent-bench","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_evalawarebench_6e7fbcdf","familyId":"bmf_3a8c0d6f9044","name":"EvalAwareBench","oneLine":"EvalAwareBench is a factor-controlled benchmark of 100 paired safety-capability tasks, each toggling one of eight evaluation-awareness triggers while holding the underlying request fixed. It measures recognition and behavioral change in language models under evaluation.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.23055","pdf":"https://arxiv.org/pdf/2605.23055","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.23055"},"evidence":{"snippet":"To study which factors each model is sensitive to and how they interact, we propose \\textbf{EvalAwareBench}, a factor-controlled benchmark of 100 paired safety-capability tasks where each of the eight factors can be independently toggled, varying evaluative signals while holding the underlying request fixed.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.23055"},"ranking":{},"description":"EvalAwareBench is a factor-controlled benchmark of 100 paired safety-capability tasks, each toggling one of eight evaluation-awareness triggers while holding the underlying request fixed. It measures recognition and behavioral change in language models under evaluation.","whyItMatters":"Evaluation awareness can distort benchmark results, especially for safety evaluations. EvalAwareBench enables controlled measurement of model sensitivity to evaluative signals, helping researchers identify and mitigate threats to benchmark validity.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e8ec7897967b2d9aa9d0c4f85ae6c9690761e4b38b53f85bf00fb2685dc30bb3"},"motivation":"Frontier language models sometimes recognize that they are being evaluated and adjust their behavior, undermining validity of benchmark results.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.23055","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_evallab_50a8738a","familyId":"bmf_5ed793692e46","name":"EvalLab","oneLine":"Provides a frozen 120-task Python coding benchmark with paired comparisons and uncertainty analysis focused on ranking stability.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/bagheri365/evallab","pdf":null,"project":null,"code":"https://github.com/bagheri365/evallab","data":null,"hfPaper":null},"evidence":{"snippet":"evallab Research-style evaluation of local LLMs focused on paired comparisons, uncertainty, benchmark stability, and trustworthy model-ranking conclusions.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:bagheri365/evallab"},"ranking":{"30d":{"score":23,"rank":124,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":328,"coverage":0.55,"confidence":"Low"}},"description":"Provides a frozen 120-task Python coding benchmark with paired comparisons and uncertainty analysis focused on ranking stability.","whyItMatters":"Studies how benchmark composition and ambiguity affect model-ranking trustworthiness, informing evaluation methodology.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"ce71816c1de976ddb9fcee59a921f31d8f9ba389963eab1cb8fca928ebba8e37"},"motivation":"evallab Research-style evaluation of local LLMs focused on paired comparisons, uncertainty, benchmark stability, and trustworthy model-ranking conclusions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The project presents a research-style evaluation with a clear benchmark but lacks an independent paper or external adoption; its named status as a benchmark is self-declared and early-stage."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/bagheri365/evallab","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":20,"confidence":"Low","horizon":"7d","reason":"The meta-evaluation focus and small task set limit immediate interest, and the project currently exists only as a personal repository without formal publication."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_dfdf41ea3061f83c","familyId":"catalog_family_dfdf41ea3061f83c","name":"EvalPlus","oneLine":"A rigorous code synthesis evaluation framework that augments existing datasets with extensive test cases generated by LLM and mutation-based strategies to better assess functional correctness of generated code, including HumanEval+ with 80x more test cases","description":"A rigorous code synthesis evaluation framework that augments existing datasets with extensive test cases generated by LLM and mutation-based strategies to better assess functional correctness of generated code, including HumanEval+ with 80x more test cases","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/evalplus","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dfdf41ea3061f83c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/evalplus"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"evalplus","url":"https://llm-stats.com/benchmarks/evalplus","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","code"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_event-activitynet_532df77c","familyId":"bmf_f0108426e128","name":"Event ActivityNet","oneLine":"Event ActivityNet is a large-scale simulated-event benchmark for untrimmed action understanding, derived from ActivityNet videos. It includes 3,263 videos, 200 action classes, event-voxel representations, temporal annotations, and supports recognition, event-language alignment, and online temporal localization.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.01948","pdf":"https://arxiv.org/pdf/2608.01948","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01948"},"evidence":{"snippet":"We introduce Event ActivityNet, a large-scale simulated-event benchmark derived from human-annotated, untrimmed ActivityNet videos.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01948"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Event ActivityNet is a large-scale simulated-event benchmark for untrimmed action understanding, derived from ActivityNet videos. It includes 3,263 videos, 200 action classes, event-voxel representations, temporal annotations, and supports recognition, event-language alignment, and online temporal localization.","whyItMatters":"Existing datasets lack long-horizon event-based understanding. Event ActivityNet provides a scalable benchmark for long-horizon event modeling with multiple tasks, enabling evaluation of models on untrimmed action understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"07dbd75f826bdb9f61f3bc2b6b6b25b3f322c60be70ce8e799bd18c9ccfb5e47"},"motivation":"Long-horizon event-based action understanding remains underexplored because existing datasets largely comprise short, trimmed clips, while collecting native event streams with dense temporal annotations is costly.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.01948","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_evid-bench_e9fc3b86","familyId":"bmf_79757df46833","name":"EVID-Bench","oneLine":"EVID-Bench is a benchmark for search-grounded video misinformation detection, covering 222 videos and multiple manipulation types.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04098","pdf":"https://arxiv.org/pdf/2606.04098","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04098"},"evidence":{"snippet":"We introduce \\textbf{EVID-Bench}, a benchmark for search-grounded video misinformation detection, where a system must search the open web for related videos and identify what information is false through cross-video comparison.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04098"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"EVID-Bench is a benchmark for search-grounded video misinformation detection, covering 222 videos and multiple manipulation types.","whyItMatters":"It highlights challenges in detecting video misinformation that requires external evidence, informing verification systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ef30a0c3b20ebc3419c8b670dcb196c5ddfad62ca2662af0b98b958a5b5e9822"},"motivation":"Video misinformation increasingly operates at the semantic and evidential level: authentic footage may be selectively edited, temporally reordered, spliced across sources, or augmented with AI-generated content to construct false narratives.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04098","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_evo-bench_a13a17e9","familyId":"bmf_d555691c91f5","name":"Evo-Bench","oneLine":"Evo-Bench is a benchmark for evaluating whether language models can autonomously improve their agent harness, with 608 harness-sensitive tasks across Search, Office, and General domains, fixed policy model, and resource budget.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09096","pdf":"https://arxiv.org/pdf/2608.09096","project":null,"code":"https://github.com/RUCAIBox/Evo-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.09096"},"evidence":{"snippet":"To address these challenges, we introduce Evo-Bench, the first benchmark designed to evaluate models' intrinsic harness-evolving capabilities across Search, Office, and General agent domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":18,"hfDailySubmittedAt":"2026-08-11T00:00:00.000Z","githubStars":11,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09096"},"ranking":{"30d":{"score":44,"rank":43,"coverage":0.85,"confidence":"High"},"90d":{"score":41,"rank":139,"coverage":0.7,"confidence":"Medium"}},"description":"Evo-Bench is a benchmark for evaluating whether language models can autonomously improve their agent harness, with 608 harness-sensitive tasks across Search, Office, and General domains, fixed policy model, and resource budget.","whyItMatters":"It isolates harness evolution from base model strength, addressing a gap in evaluating agent self-improvement capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"84ac3298bdc134f6072c77d653b047219d85d1f68801c26089d65708bb1e5d62"},"motivation":"Large Language Models (LLMs) have driven rapid progress in autonomous agents, yet standard evaluations remain confined to static task solving.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09096","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"RUCAIBox","organizationType":"academic-lab","sourceUrl":"https://github.com/RUCAIBox/Evo-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_evoagentbench_cbe93d38","familyId":"bmf_3e6250cd3257","name":"EvoAgentBench","oneLine":"EvoAgentBench evaluates agent self-evolution via ability transfer across web research, algorithmic reasoning, software engineering, and knowledge work, using trace-grounded abilities and domain graphs with train/test splits.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Reasoning","Factuality"],"topics":["Self-Evolution","Agents","Code"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Inspectable","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.05202","pdf":"https://arxiv.org/pdf/2607.05202","project":null,"code":null,"data":"https://huggingface.co/datasets/EverMind-AI/EvoAgentBench","hfPaper":"https://huggingface.co/papers/2607.05202"},"evidence":{"snippet":"We introduce EvoAgentBench, a benchmark for agent self-evolution via Ability-guided transfer across four agentic domains: web research, algorithmic reasoning, software engineering, and knowledge work.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":314,"hfDatasetLikes":16},"source":{"type":"arxiv","id":"2607.05202"},"ranking":{"90d":{"score":44,"rank":123,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":30,"datasetRankPopulation":66}},"description":"EvoAgentBench evaluates agent self-evolution via ability transfer across web research, algorithmic reasoning, software engineering, and knowledge work, using trace-grounded abilities and domain graphs with train/test splits.","whyItMatters":"Current evaluations don't isolate procedural reuse. This benchmark enables fine-grained diagnosis of experience encoding, routing, and uptake in agent self-evolution.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c159f16a4c3d63f272f17bea01e82196c6399f357ea0c0e5bd90a169b8e170f7"},"motivation":"Agent self-evolution in long-horizon LLM systems is largely procedural: useful experience is not merely stored information, but reusable procedures for searching, debugging, and verification.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.05202","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_evoarena_70b33a0d","familyId":"bmf_75e1fa5bb7d6","name":"EvoArena","oneLine":"EvoArena is a benchmark suite for LLM agents in dynamic environments, covering terminal workflows, software repositories, and social preferences. It evaluates step and chain accuracy under progressive environment updates.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13681","pdf":"https://arxiv.org/pdf/2606.13681","project":null,"code":"https://github.com/Aiden0526/EvoArena","data":null,"hfPaper":"https://huggingface.co/papers/2606.13681"},"evidence":{"snippet":"To address this gap, we introduce EvoArena, a benchmark suite that models environment changes as sequences of progressive updates across terminal, software, and social domains.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":143,"hfDailySubmittedAt":"2026-06-12T00:00:00.000Z","githubStars":23,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13681"},"ranking":{"90d":{"score":52,"rank":55,"coverage":0.7,"confidence":"Medium"}},"description":"EvoArena is a benchmark suite for LLM agents in dynamic environments, covering terminal workflows, software repositories, and social preferences. It evaluates step and chain accuracy under progressive environment updates.","whyItMatters":"Real-world deployments are dynamic, and existing benchmarks often ignore environment evolution. EvoArena provides a protocol for evaluating agent adaptation and highlights the need for memory models that track changes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b867bf3a4511c256aa422e7d59d8b6e44ef587e04b2a370e8d4b6bb124f6271c"},"motivation":"Large language model (LLM) agents have achieved strong performance on a wide range of benchmarks, yet most evaluations assume static environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13681","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"EvoArena Project","organizationType":"community","sourceUrl":"https://github.com/Aiden0526/EvoArena","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_evobrowsecomp_ef2e05ce","familyId":"bmf_482f8121203a","name":"EvoBrowseComp","oneLine":"EvoBrowseComp evaluates search agents on evolving knowledge with 800 contamination-free questions synthesized from live web traversal. The benchmark is designed to be regularly updated to prevent contamination.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13120","pdf":"https://arxiv.org/pdf/2606.13120","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.13120"},"evidence":{"snippet":"In this paper, we introduce EvoBrowseComp, an evolving benchmark of 400 English and 400 Chinese contamination-free complex questions synthesized via live-web traversal.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":4,"hfDailySubmittedAt":"2026-06-12T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13120"},"ranking":{"90d":{"score":47,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"EvoBrowseComp evaluates search agents on evolving knowledge with 800 contamination-free questions synthesized from live web traversal. The benchmark is designed to be regularly updated to prevent contamination.","whyItMatters":"Static benchmarks suffer from contamination and memorization, obscuring genuine retrieval. EvoBrowseComp offers a dynamic, auto-updated evaluation, which is critical for measuring true browsing competence.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"15444e18d15747256fc0896ef09365d4f74f1c087ba5ad1c72b655a0b2ccacd6"},"motivation":"Search Agents -- large language models augmented with search tools -- have intensified the need for future-proof evaluation benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13120","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_evoclawbench_03620568","familyId":"bmf_3bf0a17e8b5d","name":"EvoClawBench","oneLine":"EvoClawBench evaluates whether an agent runtime can convert evidence from its own runs into reusable skills that improve fresh executions. It covers 100 tasks and 502 sub-problems across coding, data, office, security, operations, and domain-document workflows, comparing direct execution, pre-authored skills, and post-run skill summarization.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.09711","pdf":"https://arxiv.org/pdf/2607.09711","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.09711"},"evidence":{"snippet":"We introduce EvoClawBench, a benchmark for this closed-loop skill-learning question on repeated, fixture-backed tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.09711"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"EvoClawBench evaluates whether an agent runtime can convert evidence from its own runs into reusable skills that improve fresh executions. It covers 100 tasks and 502 sub-problems across coding, data, office, security, operations, and domain-document workflows, comparing direct execution, pre-authored skills, and post-run skill summarization.","whyItMatters":"The evaluation gap is assessing closed-loop skill learning in agents, where benefits are selective and cost-sensitive rather than automatic. The benchmark provides a decision value for runtime developers and users considering skill-authoring loops.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"245f3c84838aebaf1465c81f18340a4346f4ad460b6505d9bda65fc4b1e648ec"},"motivation":"Existing agent benchmarks primarily test task completion, tool use, or skill utility, but do not isolate whether a runtime can convert evidence from its own runs into reusable skills that improve fresh executions after authoring overhead.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.09711","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_evocode-bench_60a115e8","familyId":"bmf_7accae8d1fc6","name":"EvoCode-Bench","oneLine":"EvoCode-Bench evaluates coding agents in multi-turn iterative interactions, with 26 stateful coding tasks and 227 evaluated rounds. Each task preserves the agent's workspace for 5-15 rounds, specifies requirements via observable behavior, and uses cumulative executable tests to verify new and prior requirements. Scoring uses MT@4, a four-attempt fail-stop multi-round score, and SR, a single-round score from a reference-completed prior state.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24110","pdf":"https://arxiv.org/pdf/2605.24110","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24110"},"evidence":{"snippet":"We introduce EvoCode-Bench, a benchmark of 26 stateful coding tasks and 227 evaluated rounds.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24110"},"ranking":{},"description":"EvoCode-Bench evaluates coding agents in multi-turn iterative interactions, with 26 stateful coding tasks and 227 evaluated rounds. Each task preserves the agent's workspace for 5-15 rounds, specifies requirements via observable behavior, and uses cumulative executable tests to verify new and prior requirements. Scoring uses MT@4, a four-attempt fail-stop multi-round score, and SR, a single-round score from a reference-completed prior state.","whyItMatters":"Existing coding benchmarks typically evaluate a single specification and final output, missing the ability to handle evolving requirements. EvoCode-Bench fills this gap by assessing whether agents can maintain a working codebase over multiple rounds. The gap between SR and MT@4 scores reveals that high single-turn performance does not guarantee sustained multi-turn success, providing a more relevant evaluation for real-world iterative development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6c10356ec27b25e4d29c0d964ed27fd97db09167b3f369ba416cb2b21b39da24"},"motivation":"Coding agents are increasingly used as iterative development partners, but most benchmarks still evaluate one specification followed by one final assessment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24110","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Harbor","organizationType":"community","sourceUrl":"https://huggingface.co/papers/2605.24110","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_evoharmbench-breaking-content-moderation-w_fad69c38","familyId":"bmf_b9d84094b186","name":"EvoHarmBench","oneLine":"Dynamic adversarial evaluation framework that evolves evasion strategies across 229 semantic sub-clusters from 5,002 real-world samples to assess content moderation systems.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.27844","pdf":"https://arxiv.org/pdf/2608.27844","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To the best of our knowledge, we present EvoHarmBench, the first dynamic adversarial evaluation framework for content moderation systems.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence","named entity is a model, method, framework, or dataset used in evaluation"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27844"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Dynamic adversarial evaluation framework that evolves evasion strategies across 229 semantic sub-clusters from 5,002 real-world samples to assess content moderation systems.","whyItMatters":"Addresses the gap between static benchmarks and online moderation by providing an iterative evaluation protocol with public code promised, supporting more robust safety testing.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T04:51:30.440525Z","inputHash":"985c4b793b3fcfcaae53d45d5492deff8cdc139254d89e9aa71ab46116df649f"},"motivation":"Existing evaluations of harmful content detection rely predominantly on static benchmarks, which struggle to reflect the interactive adversarial ecosystem of real-world content platforms where users continuously revise their expressions in response to moderation feedback.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T04:51:30.440525Z","model":"deepseek-v4-pro","decisionReason":"Paper announces full release of benchmark data, evaluation framework, and code, with clear scoring over iterative optimization.","canonicalNameSource":"abstract","canonicalNameEvidence":"we present EvoHarmBench, the first dynamic adversarial evaluation framework for content moderation systems."},"publication":{"status":"acceptance_claimed","venue":"Findings of EMNLP 2026","evidence":"Accepted to the Findings of EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2608.27844","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T01:03:30.163531Z"},"venueAttempts":[{"venueName":"Findings of EMNLP 2026","reviewStatus":"accepted","decisionRaw":"Accepted to the Findings of EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.27844","observedAt":"2026-08-31T01:03:30.163531Z","rawValue":"Accepted to the Findings of EMNLP 2026","level":"author-claim"}]}],"attentionForecast":{"score":82,"confidence":"Medium","horizon":"7d","reason":"Timely content moderation evaluation with dynamic adversarial settings and EMNLP Findings acceptance, likely to attract significant interest."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_evopolicygym_6d1d11a9","familyId":"bmf_887be45686cc","name":"EvoPolicyGym","oneLine":"EvoPolicyGym evaluates autonomous policy evolution in interactive RL environments. A harness-model agent iteratively edits an executable policy under a fixed interaction budget. The benchmark records trajectories of programs, submissions, feedback, and selection, and scores agents on held-out episodes.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.02440","pdf":"https://arxiv.org/pdf/2607.02440","project":null,"code":"https://github.com/Linzwcs/EvoPolicyGym","data":null,"hfPaper":"https://huggingface.co/papers/2607.02440"},"evidence":{"snippet":"We instantiate this setting in EvoPolicyGym, a benchmark built from compact interactive RL environments that evaluates how agents iteratively improve explored policies.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":51,"hfDailySubmittedAt":"2026-07-03T00:00:00.000Z","githubStars":210,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02440"},"ranking":{"90d":{"score":66,"rank":10,"coverage":0.7,"confidence":"Medium"}},"description":"EvoPolicyGym evaluates autonomous policy evolution in interactive RL environments. A harness-model agent iteratively edits an executable policy under a fixed interaction budget. The benchmark records trajectories of programs, submissions, feedback, and selection, and scores agents on held-out episodes.","whyItMatters":"Existing evaluations often collapse iterative improvement into a final score or confound it with software-engineering progress. EvoPolicyGym isolates the capability to improve policies from bounded feedback, providing trajectory-level diagnostics that distinguish how agents allocate budget and refine policies. This supports comparison of agents on a controlled, reusable protocol.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a657c35593f2b602c149c383fb8440215e4e327a769b92ed67e2939bb4795725"},"motivation":"Autonomous agents are increasingly expected to improve executable policies through feedback, yet existing evaluations often collapse this process into a final score or confound it with open-ended software-engineering progress.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02440","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"EvoPolicyGym Team","organizationType":"academic-lab","sourceUrl":"https://github.com/Linzwcs/EvoPolicyGym","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_evoshift-bench_8a39729f","familyId":"bmf_bca2abf188a6","name":"EvoShift-Bench","oneLine":"Evaluates continual visual learning under evolving semantic concept shift using ImageNet, iNaturalist, CUB-200-2011, and DomainNet with semantic transitions and metrics such as Rewrite Accuracy, Preservation Accuracy, Obsolete Retention, and Selective Revision Score.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-26","recognitionConfidence":0.65,"links":{"report":"https://arxiv.org/abs/2608.23903","pdf":"https://arxiv.org/pdf/2608.23903","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We further introduce EvoShift-Bench, spanning ImageNet, iNaturalist, CUB-200-2011, and DomainNet, with semantic transitions including class split, merge, boundary revision, insertion, partial redefinition, recurrence, and mixed semantic--appearance shift.","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23903"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates continual visual learning under evolving semantic concept shift using ImageNet, iNaturalist, CUB-200-2011, and DomainNet with semantic transitions and metrics such as Rewrite Accuracy, Preservation Accuracy, Obsolete Retention, and Selective Revision Score.","whyItMatters":"Addresses the gap of evolving semantic concepts in long-lived visual systems, providing a benchmark to assess selective semantic revision and knowledge preservation.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"449e05ca5c996e76215e9aa845fbedfa2b4b95f95172ea760493794b1973014c"},"motivation":"Visual foundation models are commonly adapted under the assumption that the appearance of incoming data may change while the semantic meaning of the prediction task remains fixed.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The paper formally introduces EvoShift-Bench with a repeatable evaluation object and comparable scoring contract, and the abstract indicates a public path through the paper and arxiv link.","canonicalNameSource":"abstract","canonicalNameEvidence":"We further introduce EvoShift-Bench, spanning ImageNet, iNaturalist, CUB-200-2011, and DomainNet"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.23903","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses a timely continual learning challenge with broad dataset coverage, though limited artifact evidence may reduce early visibility."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_exe-bench_d4e86d18","familyId":"bmf_e34bb0462558","name":"EXE-Bench","oneLine":"EXE-Bench evaluates AI-based Windows malware detectors on performance, temporal robustness, adversarial robustness, and computational overhead, aggregated into a single score.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.24177","pdf":"https://arxiv.org/pdf/2607.24177","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.24177"},"evidence":{"snippet":"For these reasons, we develop EXE-Bench, a comprehensive benchmark of AI-based Windows malware detectors.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24177"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"EXE-Bench evaluates AI-based Windows malware detectors on performance, temporal robustness, adversarial robustness, and computational overhead, aggregated into a single score.","whyItMatters":"Provides a comprehensive benchmark to guide deployment decisions, highlighting tradeoffs between feature-engineered and deep network detectors.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1fce9ab62c0e02381b7a39866eebc53cdedc67cc162a1d3c94932f8f1eeddb1b"},"motivation":"Due to the lack of systematic evaluations, we are not yet able to determine which AI-based Windows malware detector to deploy in production, since existing evaluations (i) differ in terms of data used for both training and testing; (ii) do not consider temporal analysis to showcase whether models withstand the passage of time; (iii) avoid security evaluations with adversarial attacks that could highlight their britt…","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24177","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_exphy-a-benchmark-for-explicit-physical-pr_ac5ba09f","familyId":"bmf_1779aacd1819","name":"ExPhy","oneLine":"24,000 simulated scenes for multi-object trajectory forecasting with explicit labels for mass, friction, and restitution, including ID and OOD splits over physical parameters and initial states.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-21","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.20009","pdf":"https://arxiv.org/pdf/2608.20009","project":null,"code":"https://github.com/Zest86/ExPhy","data":null,"hfPaper":null},"evidence":{"snippet":"To address this gap, we introduce \\emph{ExPhy}, a multi-object trajectory forecasting benchmark containing 24,000 simulated physical scenes with explicit object-level labels for mass, friction, and restitution.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.20009"},"ranking":{"30d":{"score":33,"rank":133,"coverage":0.55,"confidence":"Low"},"90d":{"score":29,"rank":346,"coverage":0.55,"confidence":"Low"}},"description":"24,000 simulated scenes for multi-object trajectory forecasting with explicit labels for mass, friction, and restitution, including ID and OOD splits over physical parameters and initial states.","whyItMatters":"Enables joint evaluation of trajectory forecasting and physical property estimation, revealing that accurate forecasts do not guarantee correct physical understanding.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"a25cfacd150c6c3ba74a85bae4812c3cdde3401eaa8774be6430686559d7f76c"},"motivation":"Understanding object dynamics requires not only predicting future trajectories but also examining whether a model captures the physical properties that govern motion.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The paper clearly defines the benchmark and provides a GitHub repository with code and data, ensuring reproducibility and public reuse.","canonicalNameSource":"paper_title","canonicalNameEvidence":"ExPhy: A Benchmark for Explicit Physical Property Learning in Multi-Object Trajectory Forecasting"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.20009","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-21T04:30:09.569171Z"},"attentionForecast":{"score":40,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses an emerging area of physics-guided learning and provides open code and data, likely to receive focused attention from trajectory forecasting researchers."},"evaluationMode":"public_reusable","publishers":[{"name":"Beijing University of Posts and Telecommunications","organizationType":"academic-lab","sourceUrl":"https://github.com/Zest86/ExPhy","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_explainbench_0a1efae5","familyId":"bmf_92c6f529f527","name":"ExplainBench","oneLine":"Evaluates code explanations from agents by checking whether explanations enable an LLM to correctly answer questions about intended behavior and patch effects, using a question-based suite derived from agent patches.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.26451","pdf":"https://arxiv.org/pdf/2607.26451","project":null,"code":"https://github.com/explainbench/explainbench-cli","data":null,"hfPaper":"https://huggingface.co/papers/2607.26451"},"evidence":{"snippet":"To bridge this gap, we propose ExplainBench, a benchmark to automatically evaluate explanations from coding agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":13,"hfDailySubmittedAt":"2026-08-05T00:00:00.000Z","githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26451"},"ranking":{"90d":{"score":22,"rank":385,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates code explanations from agents by checking whether explanations enable an LLM to correctly answer questions about intended behavior and patch effects, using a question-based suite derived from agent patches.","whyItMatters":"Agent explanations are often untrusted; this benchmark provides a quantitative way to compare explanation quality across agents, which is not captured by existing code-generation benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5fe6f8a58f7fab3118735bf4061feeb81ed2645e1b3f97274ebc178c70a356bb"},"motivation":"Large Language Model (LLM) agents have seen rapid adoption in software engineering.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26451","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_19fc6e81d91e8949","familyId":"catalog_family_19fc6e81d91e8949","name":"ExploitBench","oneLine":"ExploitBench is a cybersecurity benchmark that evaluates a model's ability to discover and exploit software vulnerabilities, reported as the fraction of challenges where the model captures the target (Cap%).","description":"ExploitBench is a cybersecurity benchmark that evaluates a model's ability to discover and exploit software vulnerabilities, reported as the fraction of challenges where the model captures the target (Cap%).","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External","Safety","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://exploitbench.ai/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_19fc6e81d91e8949"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/exploitbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/exploitbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"exploitBench","url":"https://benchlm.ai/benchmarks/exploitbench","paperUrl":"https://exploitbench.ai/","year":"2026","fullName":"ExploitBench v8-bench","format":"Capability coverage percentage over 16 flags","tasks":"V8 exploit synthesis runs","successorKey":null},{"catalog":"llm-stats","sourceId":"exploitbench","url":"https://llm-stats.com/benchmarks/exploitbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["external","safety","agents","code"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_f0c046f65c15f6b8","familyId":"catalog_family_f0c046f65c15f6b8","name":"ExploitGym","oneLine":"ExploitGym is a large-scale, realistic benchmark built from real-world vulnerabilities across userspace programs, Google's V8 engine, and the Linux kernel. Given a vulnerability and a proof-of-vulnerability input, agents must craft a working end-to-end exploit that achieves unauthorized code execution.","description":"ExploitGym is a large-scale, realistic benchmark built from real-world vulnerabilities across userspace programs, Google's V8 engine, and the Linux kernel. Given a vulnerability and a proof-of-vulnerability input, agents must craft a working end-to-end exploit that achieves unauthorized code execution.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Safety","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2605.11086","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f0c046f65c15f6b8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/exploitgym"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/exploitgym"}],"catalogSources":[{"catalog":"benchlm","sourceId":"exploitGym","url":"https://benchlm.ai/benchmarks/exploitgym","paperUrl":"https://arxiv.org/abs/2605.11086","year":"2026","fullName":"ExploitGym","format":"Working exploit generation","tasks":"898 exploitation tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"exploitgym","url":"https://llm-stats.com/benchmarks/exploitgym","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","safety","agents","code"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_expomotion_345f5881","familyId":"bmf_3b0426612bd6","name":"ExpoMotion","oneLine":"ExpoMotion is a large-scale benchmark for multi-exposure fusion with dynamic scenes, containing 1,738 sequences and 10,909 images across diverse environments. It provides high-fidelity ground truth for reference-based evaluation and a separate set for no-reference evaluation. The benchmark includes training and testing splits with controlled and real-world motions.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.03110","pdf":"https://arxiv.org/pdf/2607.03110","project":null,"code":"https://github.com/Leo-LiuYao/ExpoMotion","data":null,"hfPaper":"https://huggingface.co/papers/2607.03110"},"evidence":{"snippet":"In response, we introduce ExpoMotion, a large-scale benchmark designed to evaluate deghosting capabilities.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.03110"},"ranking":{"90d":{"score":33,"rank":220,"coverage":0.55,"confidence":"Low"}},"description":"ExpoMotion is a large-scale benchmark for multi-exposure fusion with dynamic scenes, containing 1,738 sequences and 10,909 images across diverse environments. It provides high-fidelity ground truth for reference-based evaluation and a separate set for no-reference evaluation. The benchmark includes training and testing splits with controlled and real-world motions.","whyItMatters":"Existing multi-exposure fusion benchmarks often neglect dynamic scenes and lack reliable ground truth, hindering evaluation of deghosting capabilities. ExpoMotion addresses this gap by providing a large-scale dataset with high-quality ground truth, enabling reproducible comparison of methods that handle motion-induced artifacts. This supports practical deployment in real-world scenarios where motion is common.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"13d77869c5a256919370d81a4ccbc50f4705a49fc28cd32745042ea863e6c26e"},"motivation":"Multi-Exposure Fusion (MEF) effectively extends dynamic range, but practical deployment is hindered by motion-induced ghosting and the scarcity of high-quality dynamic benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026","evidence":"Accepted by ECCV 2026","evidenceUrl":"https://arxiv.org/abs/2607.03110","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026","reviewStatus":"accepted","decisionRaw":"Accepted by ECCV 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.03110","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by ECCV 2026","level":"author-claim"}]}],"publishers":[{"name":"ExpoMotion Project","organizationType":"academic-lab","sourceUrl":"https://github.com/Leo-LiuYao/ExpoMotion","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_extractbench_2ecdf546","familyId":"bmf_b7c261db99d3","name":"ExtractBench","oneLine":"ExtractBench evaluates schema-guided extraction from enterprise documents. Given a document and a user-defined JSON schema, systems must return schema-valid JSON with correct values, include every record of repeated structures, mark missing fields as null, and provide source evidence. The benchmark includes 4,869 pages across 370 documents, 8 business domains, and 67 document types, with tags for challenge, perception, table structure, length, and domain. Scoring uses unified value F1 for value accuracy and word- and page-level F1 for grounding.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.29677","pdf":"https://arxiv.org/pdf/2607.29677","project":null,"code":"https://github.com/run-llama/ExtractBench","data":"https://huggingface.co/datasets/llamaindex/ExtractBench","hfPaper":"https://huggingface.co/papers/2607.29677"},"evidence":{"snippet":"We present ExtractBench, a benchmark for schema-guided extraction and, to our knowledge, the first to score value accuracy, record completeness at scale, grounding, and measured cost together.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":24,"hfDailySubmittedAt":"2026-08-03T00:00:00.000Z","githubStars":70,"githubScope":"benchmark_repo","hfDatasetDownloads":13135,"hfDatasetLikes":25},"source":{"type":"arxiv","id":"2607.29677"},"ranking":{"90d":{"score":64,"rank":15,"coverage":1.0,"confidence":"High","datasetDownloadRank":2,"datasetRankPopulation":66}},"description":"ExtractBench evaluates schema-guided extraction from enterprise documents. Given a document and a user-defined JSON schema, systems must return schema-valid JSON with correct values, include every record of repeated structures, mark missing fields as null, and provide source evidence. The benchmark includes 4,869 pages across 370 documents, 8 business domains, and 67 document types, with tags for challenge, perception, table structure, length, and domain. Scoring uses unified value F1 for value accuracy and word- and page-level F1 for grounding.","whyItMatters":"Enterprise workflows increasingly rely on agents for schema-guided extraction, where errors can lead to wrong payments or decisions. ExtractBench addresses the lack of a benchmark that jointly measures value accuracy, completeness, grounding, and cost, providing a standardized evaluation for comparing extraction systems on realistic document types and lengths.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"89d3015ed6801258917c496c948fb9c04f09e17fbe7b0257ca0d8f949129f8fe"},"motivation":"Enterprise workflows increasingly rely on agents for \\emph{schema-guided extraction}: given a document and a user-defined schema, the agent faithfully follows the schema to produce the correct output with source evidence as grounding metadata.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.29677","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LlamaIndex","organizationType":"company-research-lab","sourceUrl":"https://github.com/run-llama/ExtractBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_extremewhenbench_bffe4798","familyId":"bmf_ee5095f4a257","name":"ExtremeWhenBench","oneLine":"ExtremeWhenBench evaluates natural-language temporal grounding in hour-long videos, with 2,273 open-form queries over 194 videos. Metrics include mIoU and R@k for predicted time intervals, with strict parse-failure handling.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.12300","pdf":"https://arxiv.org/pdf/2606.12300","project":null,"code":"https://github.com/naver-ai/ExtremeWhenBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.12300"},"evidence":{"snippet":"To test this, we release ExtremeWhenBench, the first open hour-scale grounding benchmark (2,273 queries over 194 videos, mean 75.7 min, max 9 hr) with an open-form query distribution.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":8,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.12300"},"ranking":{"90d":{"score":33,"rank":210,"coverage":0.7,"confidence":"Medium"}},"description":"ExtremeWhenBench evaluates natural-language temporal grounding in hour-long videos, with 2,273 open-form queries over 194 videos. Metrics include mIoU and R@k for predicted time intervals, with strict parse-failure handling.","whyItMatters":"Provides the first open hour-scale grounding benchmark to study the search problem in video understanding. Useful for developing and evaluating models for long-video retrieval and reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"84a01675c6e45d379bc29c233e8795d42c78d7afbeaed6f9da63114f6a1a4248"},"motivation":"Temporal grounding--returning the interval $[t_s, t_e]$ for a natural-language query over a video--is the language interface to long-form video, yet has been studied on short videos; the dynamics of hour-scale natural-language grounding remain underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.12300","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"NAVER AI","organizationType":"company-research-lab","sourceUrl":"https://github.com/naver-ai/ExtremeWhenBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_eyt-bench_32cb0b9c","familyId":"bmf_67a637fc68e9","name":"EYT-Bench","oneLine":"EYT-Bench evaluates multi-turn dialogue capabilities of LLMs using a decoupled three-party setup with user simulation, target modeling, and judging.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.10428","pdf":"https://arxiv.org/pdf/2607.10428","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.10428"},"evidence":{"snippet":"We introduce EYT-Bench, a human-centered benchmark whose evaluation protocol is built around a decoupled three-party design: a persona-grounded user simulator, a target model evaluated on both intent perception and response generation, and an independent, configurable ensemble of LLM judges.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.10428"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"EYT-Bench evaluates multi-turn dialogue capabilities of LLMs using a decoupled three-party setup with user simulation, target modeling, and judging.","whyItMatters":"Assessing conversational AI beyond single turns may guide development of more consistent and context-aware dialogue systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"10404f1d93198eb060bae2c1f9faa113cc8879b19b7cc5375fd9b11ece7f035a"},"motivation":"Evaluating large language models (LLMs) as multi-turn conversational partners requires probing capabilities that single-turn benchmarks miss: persona consistency, evolving intent tracking, emotional dynamics, and goal completion across many turns.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.10428","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"lib_facts_grounding","familyId":"family_facts","name":"FACTS Grounding","oneLine":"Established benchmark family · Factuality & Grounding.","area":"Factuality & Grounding","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Factuality & Grounding"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2501.03200","pdf":null,"project":"https://www.kaggle.com/facts-leaderboard","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_facts_grounding"},"ranking":{},"recordType":"family","aliases":["FACTS"],"sourceAttribution":[{"role":"benchmark-paper","url":"https://arxiv.org/abs/2501.03200"}],"adoptionRefs":["google-gemini25"],"modelReportReferences":[{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"facts-grounding","url":"https://llm-stats.com/benchmarks/facts-grounding","datasetSlug":"facts-grounding","versionCount":2,"subsetCount":1,"rowCount":null,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":["reasoning","factuality","grounding"],"catalogModelCount":13,"catalogStarCount":1},{"id":"catalog_5cf6336c4063260e","familyId":"catalog_family_5cf6336c4063260e","name":"FACTS Parametric","oneLine":"A parametric factuality benchmark reported in DeepSeek-V4 base-model evaluations.","description":"A parametric factuality benchmark reported in DeepSeek-V4 base-model evaluations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5cf6336c4063260e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/factsparametric"}],"catalogSources":[{"catalog":"benchlm","sourceId":"factsParametric","url":"https://benchlm.ai/benchmarks/factsparametric","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"FACTS Parametric","format":"Exact match","tasks":"Parametric factual recall","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0ecc6d9f488cc016","familyId":"catalog_family_0ecc6d9f488cc016","name":"Facts-VLM","oneLine":"A grounded multimodal factuality benchmark for evidence-linked answer correctness.","description":"A grounded multimodal factuality benchmark for evidence-linked answer correctness.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0ecc6d9f488cc016"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/factsvlm"}],"catalogSources":[{"catalog":"benchlm","sourceId":"factsVlm","url":"https://benchlm.ai/benchmarks/factsvlm","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"Facts-VLM","format":"Evidence-linked multimodal factuality","tasks":"Grounded factuality tasks","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_64e8f9d40676caec","familyId":"catalog_family_64e8f9d40676caec","name":"FActScore","oneLine":"A fine-grained atomic evaluation metric for factual precision in long-form text generation that breaks generated text into atomic facts and computes the percentage supported by reliable knowledge sources, with automated assessment using retrieval and language models","description":"A fine-grained atomic evaluation metric for factual precision in long-form text generation that breaks generated text into atomic facts and computes the percentage supported by reliable knowledge sources, with automated assessment using retrieval and language models","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/factscore","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_64e8f9d40676caec"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/factscore"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"factscore","url":"https://llm-stats.com/benchmarks/factscore","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_faithformbench_8ab938df","familyId":"bmf_e9cebcae140a","name":"FaithformBench","oneLine":"A benchmark for evaluating faithfulness of mathematical chain-of-thought autoformalisation, using perturbed reasoning steps to test validity and invalidity preservation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.10916","pdf":"https://arxiv.org/pdf/2608.10916","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.10916"},"evidence":{"snippet":"To address these limitations, we propose a new benchmark for AF faithfulness that is cheap to apply, sound under weak assumptions, and assesses both positive and negative examples.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10916"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark for evaluating faithfulness of mathematical chain-of-thought autoformalisation, using perturbed reasoning steps to test validity and invalidity preservation.","whyItMatters":"Addresses the need for sound and cheap evaluation of autoformalisation faithfulness, revealing sycophancy in current systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7fad17407de799e1a4cca6650362a0a22e1a56c63e10f265e61d35b348ac7f9b"},"motivation":"Autoformalisation (AF) systems map natural language reasoning steps into formal statements in a proof assistant such as Lean.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10916","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_fakei2v-bench_02c67b9d","familyId":"bmf_c3f4d16ac3a8","name":"FakeI2V-Bench","oneLine":"FakeI2V-Bench is a benchmark for evaluating image-level and video-level deepfake detectors in video detection scenarios. It comprises 97,548 videos generated by recent generation models and covers multiple content categories, with detectors scored by AUC on detection tasks.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CR"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.03096","pdf":"https://arxiv.org/pdf/2608.03096","project":null,"code":"https://github.com/CryptoAILab/FakeI2V-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.03096"},"evidence":{"snippet":"To fill this gap, we present FakeI2V-Bench, a benchmark for evaluating state-of-the-art video-level deepfake detectors in challenging scenarios, with a particular focus on systematically assessing the performance of image-level deepfake detectors in the video domain.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03096"},"ranking":{"30d":{"score":31,"rank":77,"coverage":0.55,"confidence":"Low"},"90d":{"score":31,"rank":232,"coverage":0.55,"confidence":"Low"}},"description":"FakeI2V-Bench is a benchmark for evaluating image-level and video-level deepfake detectors in video detection scenarios. It comprises 97,548 videos generated by recent generation models and covers multiple content categories, with detectors scored by AUC on detection tasks.","whyItMatters":"Existing deepfake video benchmarks lack evaluation of image-level detectors' transferability to video, and this benchmark provides a large-scale, standardized protocol to measure detector performance across both detector types, aiding in model selection and development for practical deepfake video detection.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4cbcb3e1a19c4006c5a3aa6c842acbd54bcf5a0e28aed8a1d803354198010ff9"},"motivation":"Recent advances in video generation models have significantly intensified the deepfake threat, yet the current deepfake video detection benchmarks remain underdeveloped.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"KDD 2026, Jeju, Korea, August 9-13, 2026","evidence":"To Appear in KDD 2026, Jeju, Korea, August 9-13, 2026","evidenceUrl":"https://arxiv.org/abs/2608.03096","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"KDD 2026, Jeju, Korea, August 9-13, 2026","reviewStatus":"accepted","decisionRaw":"To Appear in KDD 2026, Jeju, Korea, August 9-13, 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.03096","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"To Appear in KDD 2026, Jeju, Korea, August 9-13, 2026","level":"author-claim"}]}],"publishers":[{"name":"CryptoAILab","organizationType":"academic-lab","sourceUrl":"https://github.com/CryptoAILab/FakeI2V-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_fam-bench_1946f5cd","familyId":"bmf_07e4f08667c9","name":"FAM-Bench","oneLine":"FAM-Bench evaluates multimodal language and vision-language models on Food-as-Medicine reasoning. It includes 2500 expert-verified instances across 13 health conditions, with two tasks: dish-level suitability assessment (judging if a dish is suitable for a condition) and comparative dish analysis (ranking four dishes by suitability).","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.31410","pdf":"https://arxiv.org/pdf/2605.31410","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.31410"},"evidence":{"snippet":"We introduce FAM-Bench, a multi-modal Food-as-Medicine benchmark with 2500 nutrition-expert-verified instances across 13 diet-related health conditions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.31410"},"ranking":{},"description":"FAM-Bench evaluates multimodal language and vision-language models on Food-as-Medicine reasoning. It includes 2500 expert-verified instances across 13 health conditions, with two tasks: dish-level suitability assessment (judging if a dish is suitable for a condition) and comparative dish analysis (ranking four dishes by suitability).","whyItMatters":"The benchmark fills a gap in food AI evaluation by testing whether models can integrate ingredient, visual, and clinical nutrition constraints to make condition-aware food decisions. It provides a standardized testbed for comparing models on grounded health-aware reasoning, which is relevant for applications in nutrition guidance and chronic disease management.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c45bc785f974ced4e8d845a40772130131a79a51fff1fe234a5eec09872e3442"},"motivation":"Food-as-Medicine requires models to reason beyond what a dish is or what nutrition it contains: they must decide whether a concrete food choice is appropriate for a specific health condition.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.31410","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_fastkernels_60e561f1","familyId":"bmf_c3b50d55e8f5","name":"FastKernels","oneLine":"FastKernels evaluates GPU kernel generation for production inference across 46 representative architectures spanning 8 categories. It measures correctness and speedup of candidate kernels against reference implementations at multiple abstraction levels, from single-kernel to end-to-end serving.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation"],"topics":["Code"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.23215","pdf":"https://arxiv.org/pdf/2605.23215","project":null,"code":"https://github.com/Snowflake-AI-Research/fastkernels","data":null,"hfPaper":"https://huggingface.co/papers/2605.23215"},"evidence":{"snippet":"We introduce FastKernels, a kernel benchmark built around a minimal set of 46 representative architectures spanning 8 categories, whose kernels collectively subsume those of 96.2% (409/425) of HuggingFace Transformers architectures.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.23215"},"ranking":{},"description":"FastKernels evaluates GPU kernel generation for production inference across 46 representative architectures spanning 8 categories. It measures correctness and speedup of candidate kernels against reference implementations at multiple abstraction levels, from single-kernel to end-to-end serving.","whyItMatters":"Existing kernel benchmarks reward sandbox optimizations that fail in production. FastKernels aligns evaluation with real inference frameworks, providing a benchmark where scores translate to throughput improvements in production codebases.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c697a4fc536c397d33cd9246d6cabdeef8722d60539728d93d563042db782833"},"motivation":"LLM-based agents for GPU kernel generation are advancing rapidly, yet their progress is fundamentally constrained by the benchmarks they optimize against.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.23215","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Snowflake AI Research","organizationType":"company-research-lab","sourceUrl":"https://github.com/Snowflake-AI-Research/fastkernels","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_fault-bench_59abb402","familyId":"bmf_e1ed34f7f725","name":"FaulT-Bench","oneLine":"A benchmark of 200 troubleshooting scenarios across eight network topologies evaluating network troubleshooting LLM agents under unreliable user tickets, including false fault reports and incorrect device attribution.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.NI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.27021","pdf":"https://arxiv.org/pdf/2608.27021","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present FaulT-Bench, a benchmark of 200 troubleshooting scenarios across eight network topologies, five reimplemented from public practitioner labs, spanning genuine faults, false fault reports, incorrect device attribution, and incorrect root-cause claims.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":null,"hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27021"},"ranking":{},"description":"A benchmark of 200 troubleshooting scenarios across eight network topologies evaluating network troubleshooting LLM agents under unreliable user tickets, including false fault reports and incorrect device attribution.","whyItMatters":"Evaluates the robustness of network troubleshooting agents against noisy, real-world tickets rather than only accurate inputs.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T07:23:36.749451Z","inputHash":"94ce339ee83656aa195aeb02ee7e37e42cb0c317aff4393b4a4e13a7a34ae632"},"motivation":"LLM-based agents are increasingly proposed for network fault diagnosis, but existing benchmarks evaluate them only on accurate tickets and always assume a fault is present, conditions rarely met in practice.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T07:23:36.749451Z","model":"deepseek-v4-pro","decisionReason":"Paper title formally declares FaulT-Bench, describes a repeatable harness and scoring, but no code or data link is provided; publication is provisional.","canonicalNameSource":"paper_title","canonicalNameEvidence":"FaulT-Bench: Towards Benchmarking Network Troubleshooting LLM Agents under Unreliable User Tickets"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.27021","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-30T06:59:55.242506Z"},"attentionForecast":{"score":50,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses practical network troubleshooting agent robustness and is likely to attract interest from the agent evaluation community."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_fbhm_44ffb389","familyId":"bmf_752147a6df00","name":"FBHM","oneLine":"FBHM evaluates vision-language models on hateful meme detection across 25 rhetorical functionalities and 10 target communities, with 5,000 memes. Performance is measured by Macro-F1 score on this curated dataset.","area":"Vision & 3D","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.31349","pdf":"https://arxiv.org/pdf/2605.31349","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.31349"},"evidence":{"snippet":"To address this, we introduce FBHM, a systematically curated benchmark of Functionality Based Hateful Memes constructed along two orthogonal axes: 25 distinct rhetorical functionalities and 10 target communities (5,000 memes total).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.31349"},"ranking":{},"description":"FBHM evaluates vision-language models on hateful meme detection across 25 rhetorical functionalities and 10 target communities, with 5,000 memes. Performance is measured by Macro-F1 score on this curated dataset.","whyItMatters":"Existing hateful meme benchmarks confound rhetorical strategies with target community features, preventing causal evaluation of model vulnerabilities. FBHM isolates these axes, revealing that models rely on dataset-specific heuristics rather than robust reasoning, and offers a controlled environment for measuring generalization.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"65fa90f444451fd25fa576d244646aea002efee6a85f595e49f3dc0616e67931"},"motivation":"Hateful meme detection remains a formidable challenge for vision-language models, as existing benchmarks are structurally observational - confounding rhetorical hate mechanisms with target community features and preventing causal evaluation of model vulnerabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.31349","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_fedcmapss_9057f5e4","familyId":"bmf_7481bcae4988","name":"FedCMAPSS","oneLine":"Data-driven prognostics and health management has emerged as a key enabler for Industry 4.0, yet the development of robust remaining useful life (RUL) estimation models is often limited by the scarcity of run-to-failure data.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-26","firstSeenAt":"2026-08-29","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.26433","pdf":"https://arxiv.org/pdf/2608.26433","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.26433"},"evidence":{"snippet":"To address this gap, this paper introduces FedCMAPSS, a benchmark for federated RUL estimation based on the commonly-used NASA C-MAPSS dataset.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26433"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"Data-driven prognostics and health management has emerged as a key enabler for Industry 4.0, yet the development of robust remaining useful life (RUL) estimation models is often limited by the scarcity of run-to-failure data.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"acceptance_claimed","venue":"21st IEEE Conference on Industrial Electronics and Applications (ICIEA 2026)","evidence":"Accepted at the 21st IEEE Conference on Industrial Electronics and Applications (ICIEA 2026)","evidenceUrl":"https://arxiv.org/abs/2608.26433","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"21st IEEE Conference on Industrial Electronics and Applications (ICIEA 2026)","reviewStatus":"accepted","decisionRaw":"Accepted at the 21st IEEE Conference on Industrial Electronics and Applications (ICIEA 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.26433","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at the 21st IEEE Conference on Industrial Electronics and Applications (ICIEA 2026)","level":"author-claim"}]}],"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_fepbench_bbac7f51","familyId":"bmf_70bdd0cd8b63","name":"FEPBench","oneLine":"FEPBench evaluates text-to-image models on natural-science illustration generation using fine-grained atom set annotations, assessing instruction faithfulness, reasoning enrichment, and semantic precision across disciplines.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05949","pdf":"https://arxiv.org/pdf/2606.05949","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05949"},"evidence":{"snippet":"We introduce FEPBench, a benchmark built from carefully selected high-quality scientific illustrations across multiple disciplines and layout types.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05949"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"FEPBench evaluates text-to-image models on natural-science illustration generation using fine-grained atom set annotations, assessing instruction faithfulness, reasoning enrichment, and semantic precision across disciplines.","whyItMatters":"Existing benchmarks are holistic and miss fine-grained scientific elements. FEPBench breaks down performance by element type, revealing text-rendering and reasoning bottlenecks in state-of-the-art models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"311666389ad2f4bf6b669b5367c5145a23e7d827238756f1067ddb064a9b75d5"},"motivation":"Scientific illustrations are essential tools for communicating research findings, especially in natural science, where they visualize complex concepts and processes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05949","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_fiducia-bench_925d991c","familyId":"bmf_0e2ec9149196","name":"Fiducia-bench","oneLine":"Evaluates governability of financial agents across 626 episodes and 100 KYC/AML task variants, measuring escalation, abstention, and audit trail completeness under different agent architectures.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":0.65,"links":{"report":"https://arxiv.org/abs/2608.16055","pdf":"https://arxiv.org/pdf/2608.16055","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce Fiducia-bench, a benchmark for the governability of financial agents---whether they escalate when obligated, abstain when required, and leave an auditable trail---and use it to study a question no prior benchmark addresses: does decomposing an agent into components degrade its governance?","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.16055"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates governability of financial agents across 626 episodes and 100 KYC/AML task variants, measuring escalation, abstention, and audit trail completeness under different agent architectures.","whyItMatters":"Identifies a governance gap in decomposed agent architectures, providing a framework to assess whether policy compliance is preserved when agents are broken into components.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"4d4cd597d3065968d57509425330ef45d9f2db800eafb94af30ffb0ba60b4446"},"motivation":"Existing agent benchmarks ask whether the agent finished the task.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"Fiducia-bench is formally named, open-sourced with verification harness, and designed for comparing agent architectures on policy compliance metrics.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce Fiducia-bench, a benchmark for the governability of financial agents"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.16055","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":50,"confidence":"Medium","reason":"Governance of AI agents is a rising concern and the benchmark's open-source nature supports moderate early interest from safety and finance communities.","horizon":"7d"},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_5db7d2d8065ea818","familyId":"catalog_family_5db7d2d8065ea818","name":"FigQA","oneLine":"FigQA is a multiple-choice benchmark on interpreting scientific figures from biology papers. It evaluates dual-use biological knowledge and multimodal reasoning relevant to bioweapons development.","description":"FigQA is a multiple-choice benchmark on interpreting scientific figures from biology papers. It evaluates dual-use biological knowledge and multimodal reasoning relevant to bioweapons development.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Safety","Healthcare","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/figqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5db7d2d8065ea818"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/figqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"figqa","url":"https://llm-stats.com/benchmarks/figqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety","healthcare","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_filmbench_2191702d","familyId":"bmf_cf4912a2e3be","name":"FilmBench","oneLine":"FilmBench evaluates text-to-video and reference-to-video generation using a Cinematic Language taxonomy with 1,169 prompts and 35 sub-metrics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.24241","pdf":"https://arxiv.org/pdf/2607.24241","project":null,"code":"https://github.com/Neo-yk/FilmOps","data":null,"hfPaper":"https://huggingface.co/papers/2607.24241"},"evidence":{"snippet":"We introduce FilmBench, a text-to-video (T2V) and reference-to-video (R2V) benchmark grounded in the professional Cinematic Language of the film- academy tradition and co-developed with directors and faculty from the Beijing Film Academy and the Hujing Digital Media & Entertainment Group film studio.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":6,"hfDailySubmittedAt":null,"githubStars":31,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24241"},"ranking":{"90d":{"score":46,"rank":97,"coverage":0.7,"confidence":"Medium"}},"description":"FilmBench evaluates text-to-video and reference-to-video generation using a Cinematic Language taxonomy with 1,169 prompts and 35 sub-metrics.","whyItMatters":"Provides a film-grade benchmark with professional criteria, enabling assessment of cinematic craft beyond basic plausibility.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5823e705620fe38912c6c653d6404f6467cb393a028cf5277e9cd51b3844de10"},"motivation":"Progress in video generation keeps narrowing the visual gap between AI-generated and professionally produced footage, yet most benchmarks still draw prompts from web sources or LLM templates and score them with untrained, generic multimodal models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24241","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FilmBench team","organizationType":"academic-lab","sourceUrl":"https://github.com/Neo-yk/FilmOps","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_180c3adb07bcb406","familyId":"catalog_family_180c3adb07bcb406","name":"Finance Agent","oneLine":"Finance Agent is a benchmark for evaluating AI models on agentic financial analysis tasks, testing their ability to process financial data, perform calculations, and generate accurate analyses across various financial domains.","description":"Finance Agent is a benchmark for evaluating AI models on agentic financial analysis tasks, testing their ability to process financial data, perform calculations, and generate accurate analyses across various financial domains.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Reasoning","Finance","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/finance-agent","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_180c3adb07bcb406"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/finance-agent"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"finance-agent","url":"https://llm-stats.com/benchmarks/finance-agent","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","finance","agents"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_ffa6855012d81d65","familyId":"catalog_family_ffa6855012d81d65","name":"Finance Agent v1.1","oneLine":"Finance Agent v1.1 is an agentic financial-analysis benchmark that evaluates models on real-world finance workflows, including retrieving and reasoning over financial documents and performing multi-step calculations.","description":"Finance Agent v1.1 is an agentic financial-analysis benchmark that evaluates models on real-world finance workflows, including retrieving and reasoning over financial documents and performing multi-step calculations.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["External","Reasoning","Finance","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/finance_agent","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ffa6855012d81d65"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsfinanceagentv1"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/finance-agent-v1.1"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsFinanceAgentV1","url":"https://benchlm.ai/benchmarks/valsfinanceagentv1","paperUrl":"https://www.vals.ai/benchmarks/finance_agent","year":"2026","fullName":"Vals Finance Agent v1.1","format":"Accuracy with task-level breakdowns","tasks":"11 financial analyst task views","successorKey":null},{"catalog":"llm-stats","sourceId":"finance-agent-v1.1","url":"https://llm-stats.com/benchmarks/finance-agent-v1.1","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["external","reasoning","finance","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_2cbc1aa62e647719","familyId":"catalog_family_2cbc1aa62e647719","name":"Finance Agent v2","oneLine":"Finance Agent v2 is an agentic financial-analysis benchmark from Vals that evaluates models on real-world finance workflows, measuring their ability to retrieve and reason over financial documents, perform multi-step calculations, and produce accurate analyses.","description":"Finance Agent v2 is an agentic financial-analysis benchmark from Vals that evaluates models on real-world finance workflows, measuring their ability to retrieve and reason over financial documents, perform multi-step calculations, and produce accurate analyses.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Agentic","Reasoning","Finance","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/fabv2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2cbc1aa62e647719"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/financeagentv2"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/finance-agent-v2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"financeAgentV2","url":"https://benchlm.ai/benchmarks/financeagentv2","paperUrl":"https://www.vals.ai/benchmarks/fabv2","year":"2026","fullName":"Finance Agent v2","format":"Mean score across repeated runs","tasks":"Financial analyst task categories","successorKey":null},{"catalog":"llm-stats","sourceId":"finance-agent-v2","url":"https://llm-stats.com/benchmarks/finance-agent-v2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","finance","agents"],"catalogModelCount":26,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_6e9c868cd965c20c","familyId":"catalog_family_6e9c868cd965c20c","name":"FinanceArena","oneLine":"An AfterQuery benchmark of open-ended financial analysis that requires models to read financial data, make assumptions, and return exact answers.","description":"An AfterQuery benchmark of open-ended financial analysis that requires models to read financial data, make assumptions, and return exact answers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2501.18062","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6e9c868cd965c20c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/financearena"}],"catalogSources":[{"catalog":"benchlm","sourceId":"financeArena","url":"https://benchlm.ai/benchmarks/financearena","paperUrl":"https://arxiv.org/abs/2501.18062","year":"2025","fullName":"FinanceArena — FinanceQA Assumption-Based","format":"Open-ended financial QA with exact-match grading","tasks":"Professional financial-analysis questions","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_financial-sentiment-benchmark_ea047ce6","familyId":"bmf_1a3121aeb0c6","name":"financial-sentiment-benchmark","oneLine":"Evaluates financial news sentiment classification comparing TF-IDF, DistilBERT fine-tuned, QLoRA, and prompted GPT-5-nano on a 300-sentence test subset using macro-F1.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-21","firstSeenAt":"2026-08-22","recognitionConfidence":0.4,"links":{"report":"https://github.com/gauthamRohan/financial-sentiment-benchmark","pdf":null,"project":null,"code":"https://github.com/gauthamRohan/financial-sentiment-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"financial-sentiment-benchmark Fine-tune a small model or prompt a big one?","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:gauthamrohan/financial-sentiment-benchmark"},"ranking":{"30d":{"score":23,"rank":134,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":338,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates financial news sentiment classification comparing TF-IDF, DistilBERT fine-tuned, QLoRA, and prompted GPT-5-nano on a 300-sentence test subset using macro-F1.","whyItMatters":"Provides a seeded split and frozen test subset for repeatable comparisons of fine-tuned small models versus prompted large models on financial sentiment.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"e212801c4e1688987dd2301be511953765fb5d5cb2b37cf20bb940b68a30b6a4"},"motivation":"financial-sentiment-benchmark Fine-tune a small model or prompt a big one?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The repository provides explicit instructions and frozen evaluation but no formal benchmark name; the GitHub project is only a discovery artifact without an independent paper confirming a released benchmark."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/gauthamRohan/financial-sentiment-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-22T19:19:35.289219Z"},"attentionForecast":{"score":30,"confidence":"Low","horizon":"7d","reason":"The repository compares fine-tuned and prompted approaches with reproducible code and results, but the specialized domain and absence of a formal benchmark name limit broad circulation."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_finbalance_0aab2d5b","familyId":"bmf_5dac25fe40d5","name":"FinBalance","oneLine":"FinBalance is a multi-document accounting reconciliation benchmark built from 710 source-document bundles across eight industries, three period types, and five difficulty levels. It evaluates models on producing journal entries, balance sheets, and inconsistency labels, with a deterministic ledger for ground truth.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-06-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15949","pdf":"https://arxiv.org/pdf/2606.15949","project":null,"code":"https://github.com/Devansh1105/finbalance","data":null,"hfPaper":"https://huggingface.co/papers/2606.15949"},"evidence":{"snippet":"We introduce FinBalance, a multi-document accounting reconciliation benchmark built from source-document bundles across eight industries, three period types, and five difficulty levels.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15949"},"ranking":{"90d":{"score":23,"rank":380,"coverage":0.55,"confidence":"Low"}},"description":"FinBalance is a multi-document accounting reconciliation benchmark built from 710 source-document bundles across eight industries, three period types, and five difficulty levels. It evaluates models on producing journal entries, balance sheets, and inconsistency labels, with a deterministic ledger for ground truth.","whyItMatters":"FinBalance addresses the gap in evaluating accounting reasoning from source documents rather than prepared financial statements. It provides a reproducible and auditable benchmark for document-grounded financial reasoning, with expert validation of the design.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"171a94eddc5ceeb2b414d1545b860dda7e33b91878e40f643dd7a686f4955d38"},"motivation":"Existing financial-NLP benchmarks mostly evaluate prepared artifacts such as filings, tables, or extracted values.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15949","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FinBalance Team","organizationType":"academic-lab","sourceUrl":"https://github.com/Devansh1105/finbalance","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_finbench_34ca17f0","familyId":"bmf_5ee7524fb0ac","name":"FinBench","oneLine":"FinBench is a benchmark for evaluating calibration and uncertainty in financial forecasting with time-gated tasks, requiring probability of positive return and 80% prediction interval, scored with Brier and Winkler scores.","area":"Safety & Trustworthiness","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["stat.AP"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.16229","pdf":"https://arxiv.org/pdf/2607.16229","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.16229"},"evidence":{"snippet":"We introduce FinBench, a benchmark designed to evaluate calibration and uncertainty quality for financial forecasting in a setting that is (i) strictly time-gated to avoid look-ahead bias and (ii) evaluated with strictly proper scoring rules that penalize hallucinated confidence.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.16229"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FinBench is a benchmark for evaluating calibration and uncertainty in financial forecasting with time-gated tasks, requiring probability of positive return and 80% prediction interval, scored with Brier and Winkler scores.","whyItMatters":"Financial forecasting agents risk overconfidence; FinBench addresses this by testing probabilistic calibration under temporal constraints, but its pilot scale limits immediate comparison value.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2f20475c33eaeac878e03ec82af798089271947533fa51c38bab9656d02aeecf"},"motivation":"Large language models (LLMs) are increasingly used as components of agentic systems that observe, plan, and act.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.16229","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_finboardbench_0e7661a6","familyId":"bmf_f6dd71cc3fce","name":"FinBoardBench","oneLine":"FinBoardBench is an evaluation suite based on three financial board games: Cashflow, Acquire, and Monopoly. It assesses financial skills including personal cash flow management, corporate investment and acquisition forecasting, and competitive trade negotiations with asset auctions, using game simulations as the environment.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27896","pdf":"https://arxiv.org/pdf/2605.27896","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27896"},"evidence":{"snippet":"To bridge this gap, we present FinBoardBench, an evaluation suite based on three classic financial board games: Cashflow, Acquire, and Monopoly.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27896"},"ranking":{},"description":"FinBoardBench is an evaluation suite based on three financial board games: Cashflow, Acquire, and Monopoly. It assesses financial skills including personal cash flow management, corporate investment and acquisition forecasting, and competitive trade negotiations with asset auctions, using game simulations as the environment.","whyItMatters":"Existing static financial benchmarks do not capture dynamic decision-making in complex environments. FinBoardBench provides a repeatable protocol to measure whether LLMs can translate static reasoning into successful dynamic action, addressing a gap in financial AI evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"de9916c6d32cee47c6a8c512ac3d61c05c365793cc2b93a2a054141276c34b51"},"motivation":"Recently, large language models (LLMs) have achieved superior performance in static financial reasoning and simple dynamic trading tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27896","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_findeepindicator_6effb730","familyId":"bmf_5d885cb8e46f","name":"FinDeepIndicator","oneLine":"FinDeepIndicator evaluates deep research agents on end-to-end financial indicator construction across four stages: formula specification, data collection, indicator calculation, and answer generation. It includes 3,350 QA pairs from U.S. and Chinese markets, 10 years of historical data, and 800 listed companies.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00764","pdf":"https://arxiv.org/pdf/2608.00764","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00764"},"evidence":{"snippet":"In this work, we propose FinDeepIndicator, the first benchmark dedicated to evaluating Deep Research (DR) agents in end-to-end financial indicator construction.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00764"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"FinDeepIndicator evaluates deep research agents on end-to-end financial indicator construction across four stages: formula specification, data collection, indicator calculation, and answer generation. It includes 3,350 QA pairs from U.S. and Chinese markets, 10 years of historical data, and 800 listed companies.","whyItMatters":"Existing financial benchmarks focus on answer accuracy and assume data is provided, leaving the intermediate process of indicator construction unassessed. FinDeepIndicator bridges this gap by evaluating the full process, providing insights for building more capable and trustworthy agents in finance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4c61a3efc086d6cfbce3796c35c54e47b496117060935a889fca74fe4515706c"},"motivation":"Financial indicators are essential tools for transforming raw financial data into interpretable measures for various downstream tasks, such as valuation, risk assessment, and economic analysis.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00764","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_fined-bench_3f021fe1","familyId":"bmf_e5c0a8628ba8","name":"FinED-Bench","oneLine":"FinED-Bench evaluates error detection in financial documents across nine real-world financial scenarios, covering three cognitive complexity levels. It includes over 900 documents from 2025, with data and code publicly available.","area":"Vision & 3D","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12342","pdf":"https://arxiv.org/pdf/2608.12342","project":null,"code":"https://github.com/hedyHe/FinED-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.12342"},"evidence":{"snippet":"In this paper, we introduce \\textbf{FinED-Bench}, the first publicly \\textbf{Bench}mark for \\textbf{Fin}ancial \\textbf{E}rror \\textbf{D}etection across three levels of cognitive complexity.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12342"},"ranking":{"90d":{"score":23,"rank":382,"coverage":0.55,"confidence":"Low"}},"description":"FinED-Bench evaluates error detection in financial documents across nine real-world financial scenarios, covering three cognitive complexity levels. It includes over 900 documents from 2025, with data and code publicly available.","whyItMatters":"Financial document accuracy is critical for compliance and decision-making, yet current LLMs struggle with error detection. FinED-Bench provides a fixed protocol to assess and improve model reliability in this domain.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2da8f6491aa3abc07def0529966ba75d6325cace9d4c664e4f88fc61fa987abf"},"motivation":"Ensuring the accuracy of financial documents is critical for economic analysis, regulatory compliance, and corporate decision-making.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12342","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FinED-Bench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/hedyHe/FinED-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_finesightbench_86226917","familyId":"bmf_a5a8565d449b","name":"FineSightBench","oneLine":"FineSightBench probes fine-scale visual perception in VLMs across scales from 4 to 48 pixels, separating perception and reasoning tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07861","pdf":"https://arxiv.org/pdf/2606.07861","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07861"},"evidence":{"snippet":"As such, we introduce FineSightBench, a new benchmark that systematically probes this limit by separating perception tasks (pixel-level recognition of letters, shapes, objects) from reasoning tasks (spatial reasoning, counting, ordering over small targets) across controlled scales of 4--48px.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07861"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FineSightBench probes fine-scale visual perception in VLMs across scales from 4 to 48 pixels, separating perception and reasoning tasks.","whyItMatters":"Reveals limits in fine-grained VLM perception, motivating better evaluation, but is a probe with no public path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"98af499c020f88ea65d265f60d76a812d1785747cfe993512c319c416565cc59"},"motivation":"Recent vision-language models (VLMs) excel at multimodal understanding and reasoning, yet their fine-grained visual perception remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07861","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_finevo-bench_09755690","familyId":"bmf_46e0c5291976","name":"FinEvo-Bench","oneLine":"FinEvo-Bench evaluates self-evolving agents on 120 real-case-grounded tasks across 20 business scenes in six financial domains, with institution-provided procedures and rubrics for quality and compliance.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["Self-Evolution","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.06144","pdf":"https://arxiv.org/pdf/2608.06144","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.06144"},"evidence":{"snippet":"We introduce FinEvo-Bench, a longitudinal benchmark with 120 real-case-grounded tasks, 20 business scenes across six financial domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06144"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FinEvo-Bench evaluates self-evolving agents on 120 real-case-grounded tasks across 20 business scenes in six financial domains, with institution-provided procedures and rubrics for quality and compliance.","whyItMatters":"Most agent benchmarks treat tasks independently and cannot measure learning from experience. FinEvo-Bench provides a longitudinal evaluation that measures both professional performance and self-evolution ability, filling a gap in agent benchmarking.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a4b4a94906f9eef5f72c45bd3104639dee614322b0ec00ee21e698d1a919a589"},"motivation":"Most agent benchmarks evaluate tasks independently and cannot measure whether experience from one task helps with later tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06144","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_finevolvebench_f7d5f81e","familyId":"bmf_bdf52b68aa6a","name":"FinEvolveBench","oneLine":"FinEvolveBench evaluates self-evolving agents on low-repetition financial prediction tasks with implicit rewards, measuring utility updates over delayed returns.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["Self-Evolution","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06960","pdf":"https://arxiv.org/pdf/2606.06960","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.06960"},"evidence":{"snippet":"We introduce \\textsc{FinEvolveBench}, a benchmark for self-evolving agents on low-repetition tasks with implicit rewards.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06960"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FinEvolveBench evaluates self-evolving agents on low-repetition financial prediction tasks with implicit rewards, measuring utility updates over delayed returns.","whyItMatters":"Addresses evaluation of experience-based self-evolution under noisy feedback, relevant for agent adaptation, but lacks public comparison path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4ba5afdbdb140318753e941814557bf9a79b8ff11be109c9725c9bd594eed771"},"motivation":"Experience-based self-evolution enables language-model agents to improve their behavior by accumulating and updating experience at test time, yet existing evaluations often assume recurring task patterns and explicit success signals.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.06960","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_finexam-10k-when-retrieval-helps-financial_fa6458fb","familyId":"bmf_7cc35ac6c6c1","name":"FinExam-10K","oneLine":"Evaluates financial reasoning across CFA Levels I-III and FRM Parts I-II with 10,198 expert-reannotated questions, releasing 5,110 public items and maintaining a leaderboard on 5,088 held-out items.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":["Reasoning","Information retrieval","Factuality"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.28155","pdf":"https://arxiv.org/pdf/2608.28155","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce FinExam-10K, to our knowledge the largest reported English benchmark for this setting, with 10,198 expert-reannotated questions spanning CFA Levels I-III and FRM Parts I-II.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.28155"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates financial reasoning across CFA Levels I-III and FRM Parts I-II with 10,198 expert-reannotated questions, releasing 5,110 public items and maintaining a leaderboard on 5,088 held-out items.","whyItMatters":"Provides the first unified benchmark for CFA and FRM examinations under a single protocol, separating coverage from context-complete reasoning and enabling ongoing model comparison.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T04:51:30.440525Z","inputHash":"ba632e520c2640d94abff33d26d3364741af937561d8cabba3061f6c73fa76de"},"motivation":"Professional financial examinations require models to combine domain knowledge, calculation, and judgment, yet no benchmark covers the full CFA and FRM structure under one protocol.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T04:51:30.440525Z","model":"deepseek-v4-pro","decisionReason":"Formal benchmark with a quarterly maintained leaderboard, sequestered test set, and stated release of 5,110 questions.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce FinExam-10K, to our knowledge the largest reported English benchmark for this setting"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.28155","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T01:03:30.163531Z"},"attentionForecast":{"score":78,"confidence":"Medium","horizon":"7d","reason":"Largest financial reasoning benchmark with broad CFA/FRM coverage and a public leaderboard, likely to attract attention from the finance and NLP communities."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"specific"},{"id":"bm_finfraudbench_fc16410f","familyId":"bmf_8a9ae6e0eb11","name":"FinFraudBench","oneLine":"FinFraudBench is a heterogeneous graph benchmark for financial fraud detection. It contains two datasets (CreditCard-Fraud and BankTrans-Fraud) with up to 8.99M nodes and 89.23M directed typed edges, preserving six financial entity types and fourteen edge types. The evaluation protocol covers ranking and imbalance-sensitive classification metrics.","area":"Vision & 3D","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15177","pdf":"https://arxiv.org/pdf/2608.15177","project":"https://anonymous.4open.science/r/FinFraudBench-B002","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.15177"},"evidence":{"snippet":"To address these gaps, we present FinFraudBench, a heterogeneous graph benchmark for financial fraud detection.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15177"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FinFraudBench is a heterogeneous graph benchmark for financial fraud detection. It contains two datasets (CreditCard-Fraud and BankTrans-Fraud) with up to 8.99M nodes and 89.23M directed typed edges, preserving six financial entity types and fourteen edge types. The evaluation protocol covers ranking and imbalance-sensitive classification metrics.","whyItMatters":"Existing graph-based fraud detection benchmarks often oversimplify financial systems and lack realistic conditions. FinFraudBench provides large-scale heterogeneous graphs with natural fraud rates, enabling more realistic evaluation of fraud detection methods and comparison across models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2c048e1de37db1a521f9da04691dc7aa1f62ad85fa6d9d6255a3cf86e4592c1c"},"motivation":"The increasing complexity of digital financial systems has reshaped financial fraud detection from isolated transaction classification into relational risk reasoning over interconnected financial entities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15177","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FinFraudBench Team","organizationType":"academic-lab","sourceUrl":"https://anonymous.4open.science/r/FinFraudBench-B002","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_finguard-bench_2ed68349","familyId":"bmf_02e6f20cf325","name":"FinGuard-Bench","oneLine":"The evaluation focuses on financial regulatory non-compliance detection in LLM interactions. It involves a benchmark with expert-annotated labels at query and response levels, but the exact task, environment, or scoring setup is not detailed.","area":"Vision & 3D","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.29427","pdf":"https://arxiv.org/pdf/2605.29427","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29427"},"evidence":{"snippet":"Instantiating the pipeline on Chinese financial regulations, we release \\textbf{FinGuard-Bench}, to our knowledge the first benchmark for financial regulatory compliance detection, with expert-annotated labels at both the query and response levels.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29427"},"ranking":{},"description":"The evaluation focuses on financial regulatory non-compliance detection in LLM interactions. It involves a benchmark with expert-annotated labels at query and response levels, but the exact task, environment, or scoring setup is not detailed.","whyItMatters":"Assesses LLM compliance with financial regulations, which is critical for preventing regulatory penalties and consumer harm in financial services. The benchmark aims to measure detection capabilities across institution-specific policies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f5c8732092120b651918c492e57dd0230a19d411da15903312f2e1b3152e919b"},"motivation":"As large language models (LLMs) are increasingly deployed in financial services, a single non-compliant interaction can expose institutions to regulatory penalties and direct consumer harm.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29427","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_finixdocbench_1d95d2be","familyId":"bmf_f976d9289f2f","name":"FinixDocBench","oneLine":"A financial document parsing evaluation suite covering diverse document scenarios, but release details and public access are unclear.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":0.55,"links":{"report":"http://arxiv.org/abs/2608.22842v1","pdf":"https://arxiv.org/pdf/2608.22842v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"For evaluation, we construct FinixDocBench, a financial-domain evaluation suite covering digital-native, camera-captured, ultra-large-page, and internal-workflow scenarios, with a compliance-reviewed subset released alongside this technical report.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","named entity is a model, method, framework, or dataset used in evaluation"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22842"},"ranking":{"today":{"score":42,"rank":18,"coverage":0.65,"confidence":"Medium"},"30d":{"score":40,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"A financial document parsing evaluation suite covering diverse document scenarios, but release details and public access are unclear.","whyItMatters":"Aims to expose gaps between benchmark and deployment performance in financial document parsing, but its current availability is uncertain.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"2be65fa0758b1c63cbef68200bd7e0bb098f8024589050677477c37843f8fd19"},"motivation":"Financial document parsing requires accuracy, structural consistency, and verifiability that current benchmarks often fail to reflect.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The paper mentions a compliance-reviewed subset released but provides no evidence of public code, data, or evaluation protocol, making the benchmark difficult to access or verify.","canonicalNameSource":"abstract","canonicalNameEvidence":"we construct FinixDocBench, a financial-domain evaluation suite"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22842v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":35,"confidence":"Low","horizon":"7d","reason":"The financial domain and agentic parsing focus may draw niche interest, but low artifact readiness limits broader attention."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_finperma_25dab076","familyId":"bmf_02e40c205e8e","name":"FinPerMA","oneLine":"FinPerMA evaluates personalized memory in LLM agents using frozen longitudinal investor trajectories with theory-informed impact rules. It includes a Post-Shock checkpoint to assess integration of material events into persistent user models, with 2,994 questions from 276 personas.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04095","pdf":"https://arxiv.org/pdf/2608.04095","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.04095"},"evidence":{"snippet":"We introduce FinPerMA, an event-grounded benchmark that evaluates personalized memory against frozen longitudinal investor trajectories.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04095"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FinPerMA evaluates personalized memory in LLM agents using frozen longitudinal investor trajectories with theory-informed impact rules. It includes a Post-Shock checkpoint to assess integration of material events into persistent user models, with 2,994 questions from 276 personas.","whyItMatters":"Existing personalized-memory benchmarks lack event-driven preference adaptation. FinPerMA fills this gap by providing an event-grounded evaluation, enabling assessment of whether agents can update user models over long horizons in high-stakes financial advising contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"42a3b21e6d5f636483446b11cbadd8f9a8a5b2891695f15cae3f512dfd5bd0a8"},"motivation":"Large language model (LLM) agents are increasingly used as personalized assistants in high-stakes domains such as financial advising, yet it remains unclear whether they can maintain and update an individualized user model over long horizons.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04095","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_finpersona-bench_6543120b","familyId":"bmf_47d6e24194de","name":"FinPersona-Bench","oneLine":"Evaluates the longitudinal stability of behavioral mandates in LLM-based financial agents using a synthetic market simulation that decouples observable price from hidden fundamental value, scoring mandate adherence across calm, crash, and bubble market regimes.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.31522","pdf":"https://arxiv.org/pdf/2606.31522","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.31522"},"evidence":{"snippet":"To measure MSD objectively, we introduce FinPersona-Bench, a simulation benchmark in which a synthetic market decouples observable price from hidden fundamental value, enabling falsifiable evaluation across three failure modes: trading without signal in calm markets, panic-selling during crashes, and ignoring fundamental value during speculative bubbles.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31522"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates the longitudinal stability of behavioral mandates in LLM-based financial agents using a synthetic market simulation that decouples observable price from hidden fundamental value, scoring mandate adherence across calm, crash, and bubble market regimes.","whyItMatters":"It addresses the gap in evaluating long-horizon behavioral consistency of autonomous financial agents, providing a falsifiable method to measure mandate salience decay and informing deployment decisions for mandate-aware re-grounding strategies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4dcfae791689a136b0d677536f5d291ecc206f601cbad44aa895d432c954ac8c"},"motivation":"Large Language Models (LLMs) are increasingly deployed as autonomous financial agents initialized with explicit behavioral mandates such as \"preserve capital\" or \"avoid speculative bets\" that are meant to govern every decision throughout deployment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31522","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_finprobench_b6fc7343","familyId":"bmf_72a8e0c3d342","name":"FinProBench","oneLine":"FinProBench evaluates financial AI agents on professional tasks using rubrics derived from practitioner deliverables. It includes 1,723 curated deliverables across 57 occupations and an evaluation set of 20 tasks.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04077","pdf":"https://arxiv.org/pdf/2608.04077","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.04077"},"evidence":{"snippet":"We introduce FinProBench, a benchmark for professional financial tasks, and Role-Grounded Rubric Construction (RGRC), a reusable pipeline that derives rubrics from deliverables produced by practitioners in the same role.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04077"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FinProBench evaluates financial AI agents on professional tasks using rubrics derived from practitioner deliverables. It includes 1,723 curated deliverables across 57 occupations and an evaluation set of 20 tasks.","whyItMatters":"The benchmark addresses the need for evaluating financial AI agents against professional standards, offering an approach to derive rubrics from expert deliverables, which can reduce effort in rubric construction.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"33148f289e9f7ae843e6eaa2b9da7aa1f7aff37a902bb011593bdd36cc88e9bf"},"motivation":"Evaluating financial AI agents requires criteria aligned with real professional work.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04077","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_16edc7ea458bb8f7","familyId":"catalog_family_16edc7ea458bb8f7","name":"FinQA","oneLine":"A large-scale dataset for numerical reasoning over financial data with question-answering pairs written by financial experts, featuring complex numerical reasoning and understanding of heterogeneous representations with annotated gold reasoning programs for full explainability","description":"A large-scale dataset for numerical reasoning over financial data with question-answering pairs written by financial experts, featuring complex numerical reasoning and understanding of heterogeneous representations with annotated gold reasoning programs for full explainability","area":"Mathematical Reasoning","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning","Finance","Economics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/finqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_16edc7ea458bb8f7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/finqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"finqa","url":"https://llm-stats.com/benchmarks/finqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning","finance","economics"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"bm_finrank_f391afeb","familyId":"bmf_defe6e205daf","name":"FinRank","oneLine":"FinRank evaluates passage retrieval, reranking, and hard-negative discrimination over SEC filings, with 1185 question-answer records and curated hard negatives, scoring recall and accuracy.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":["Information retrieval"],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.07400","pdf":"https://arxiv.org/pdf/2608.07400","project":null,"code":"https://github.com/datanxt/FinRank","data":null,"hfPaper":"https://huggingface.co/papers/2608.07400"},"evidence":{"snippet":"The benchmark contains 1185 manually authored question-answer records over the 10-K and 10-Q filings of 22 companies.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07400"},"ranking":{"30d":{"score":28,"rank":98,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":268,"coverage":0.55,"confidence":"Low"}},"description":"FinRank evaluates passage retrieval, reranking, and hard-negative discrimination over SEC filings, with 1185 question-answer records and curated hard negatives, scoring recall and accuracy.","whyItMatters":"Targets provenance-sensitive financial QA, addressing the challenge of grounding answers in correct evidence, which is critical for compliance and trust in financial applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4aa060e159017ea16c219efb693d48e1d6173862880157a80bb18385e19e9292"},"motivation":"Financial question answering is typically evaluated by answer correctness, yet in SEC filings a plausible and even numerically correct answer can be grounded in the wrong evidence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07400","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FinRank Team","organizationType":"academic-lab","sourceUrl":"https://github.com/datanxt/FinRank","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"specific"},{"id":"bm_finrca-bench_2d78ed7b","familyId":"bmf_61689bc0d835","name":"FinRCA-Bench","oneLine":"FinRCA-Bench is a synthetic benchmark for financial reconciliation root-cause analysis, with 2,250 cases across 14 tables and 15 failure categories, providing evidence retrieval and reasoning evaluation.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":["Reasoning","Information retrieval"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-19","firstSeenAt":"2026-08-20","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.18534","pdf":"https://arxiv.org/pdf/2608.18534","project":null,"code":"https://github.com/PratikGhawate/FinRCA-AI-Bench","data":null,"hfPaper":null},"evidence":{"snippet":"We introduce FinRCA-Bench, a deterministic synthetic benchmark of 2,250 accounts-payable-to-bank reconciliation cases spanning 14 operational tables, including 1,500 injected failures across 15 causal categories and 750 legitimate or hard-negative cases.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.18534"},"ranking":{"30d":{"score":28,"rank":92,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":262,"coverage":0.55,"confidence":"Low"}},"description":"FinRCA-Bench is a synthetic benchmark for financial reconciliation root-cause analysis, with 2,250 cases across 14 tables and 15 failure categories, providing evidence retrieval and reasoning evaluation.","whyItMatters":"Financial AI systems need reliable evidence retrieval and reasoning, and existing benchmarks conflate the two. FinRCA-Bench separates retrieval evaluation from answer correctness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fc782ed4e27afbb14d3ccec757b5ed7ad9aa43f8750cec2b0e275d4d55ed415d"},"motivation":"Large language models are increasingly used to support financial operations, but their apparent reasoning performance can depend on whether they receive the right evidence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-20T09:45:58.504563Z","model":"deepseek-v4-flash","decisionReason":"Provides a public dataset and generation code, with clear evaluation labels and a documented protocol. The GitHub link and paper describe the benchmark's structure and use, making it reusable."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.18534","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-20T09:44:47.308583Z"},"evaluationMode":"public_reusable","publishers":[{"name":"Pratik Ghawate","organizationType":"academic-lab","sourceUrl":"https://github.com/PratikGhawate/FinRCA-AI-Bench","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"specific"},{"id":"bm_finreportbench_5d3a5df7","familyId":"bmf_33ecc6c20ee7","name":"FinReportBench","oneLine":"FinReportBench evaluates institution-grade financial report generation using 244 bilingual tasks sourced from 10,000 financial research records. It uses a 35-item rubric covering deliverability, report identity, and institutional completeness, with three judge families.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04374","pdf":"https://arxiv.org/pdf/2608.04374","project":null,"code":"https://github.com/MisterBrookT/finreportbench","data":null,"hfPaper":"https://huggingface.co/papers/2608.04374"},"evidence":{"snippet":"We introduce FinReportBench, an expert-grounded benchmark for measuring and improving institution-grade financial report generation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04374"},"ranking":{"30d":{"score":31,"rank":76,"coverage":0.55,"confidence":"Low"},"90d":{"score":31,"rank":231,"coverage":0.55,"confidence":"Low"}},"description":"FinReportBench evaluates institution-grade financial report generation using 244 bilingual tasks sourced from 10,000 financial research records. It uses a 35-item rubric covering deliverability, report identity, and institutional completeness, with three judge families.","whyItMatters":"It fills the gap in evaluating long-form financial reports for institutional delivery, providing a reliable rubric and public artifacts to measure and improve report generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1dd2d50bc647edddfc6fe7ca60b7b3687f351d68dca43bbe2d02a002bd9c4d03"},"motivation":"Large language models can produce fluent financial analysis, but fluency alone does not establish whether a report is suitable for institutional delivery.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04374","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_finriskatlas_4424cb99","familyId":"bmf_11eb05bab140","name":"FinRiskAtlas","oneLine":"Evaluates Chinese financial LLMs on static operation execution across 9,742 instances and evidence-state control via FinRisk-Ask using pre-action trajectory states.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-26","firstSeenAt":"2026-08-29","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25325","pdf":"https://arxiv.org/pdf/2608.25325","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.25325"},"evidence":{"snippet":"We introduce FinRiskAtlas, a Chinese-language benchmark that evaluates financial LLMs along two complementary dimensions: operation execution under fixed evidence states and evidence-state control under evolving review conditions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25325"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates Chinese financial LLMs on static operation execution across 9,742 instances and evidence-state control via FinRisk-Ask using pre-action trajectory states.","whyItMatters":"Addresses the gap between broad financial benchmarks and workflow-specific decisions, showing that knowledge scores do not predict operational reliability.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"15c61651b4320250333dcdecf5e8f161bf9b01fb42dcf4fad412f79e682d63ec"},"motivation":"Deploying large language models for professional financial review requires more than measuring general financial competence: models must perform the specific review operation required by a workflow and determine whether available evidence is sufficient for a defensible decision.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25325","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"attentionForecast":{"score":55,"confidence":"Low","horizon":"7d","reason":"The paper targets a professional financial domain and introduces two complementary evaluation dimensions, which may attract attention despite missing artifact links."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_d0a57b04e138eb38","familyId":"catalog_family_d0a57b04e138eb38","name":"FinSearchComp T2&T3","oneLine":"FinSearchComp T2&T3 is a combined benchmark for evaluating financial search and reasoning capabilities on Tier 2 and Tier 3 tasks, testing models' ability to retrieve and analyze complex financial information using tools.","description":"FinSearchComp T2&T3 is a combined benchmark for evaluating financial search and reasoning capabilities on Tier 2 and Tier 3 tasks, testing models' ability to retrieve and analyze complex financial information using tools.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Reasoning","Search","Finance","Economics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/finsearchcomp-t2-t3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d0a57b04e138eb38"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/finsearchcomp-t2-t3"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"finsearchcomp-t2-t3","url":"https://llm-stats.com/benchmarks/finsearchcomp-t2-t3","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","search","finance","economics"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"specific"},{"id":"catalog_38473fc736dd8308","familyId":"catalog_family_38473fc736dd8308","name":"FinSearchComp-T3","oneLine":"FinSearchComp-T3 is a benchmark for evaluating financial search and reasoning capabilities, testing models' ability to retrieve and analyze financial information using tools.","description":"FinSearchComp-T3 is a benchmark for evaluating financial search and reasoning capabilities, testing models' ability to retrieve and analyze financial information using tools.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Reasoning","Search","Finance","Economics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/finsearchcomp-t3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_38473fc736dd8308"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/finsearchcomp-t3"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"finsearchcomp-t3","url":"https://llm-stats.com/benchmarks/finsearchcomp-t3","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","search","finance","economics"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"specific"},{"id":"bm_finstressts_aacbcf42","familyId":"bmf_7cb0bbf0767d","name":"FinStressTS","oneLine":"FinStressTS is a synthetic benchmark for financial time-series forecasting with controllable mechanisms and diagnostic environments.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["q-fin.CP"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03184","pdf":"https://arxiv.org/pdf/2606.03184","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03184"},"evidence":{"snippet":"We introduce FinStressTS, a mechanism-aware synthetic benchmark that links model behavior to controlled structural causes.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03184"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FinStressTS is a synthetic benchmark for financial time-series forecasting with controllable mechanisms and diagnostic environments.","whyItMatters":"It enables failure attribution in forecasting models, improving understanding of performance under different data-generating processes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"13745d96f254df37e0763352802185611a59db530ba4e9a71d9659f0b2a5c902"},"motivation":"Financial forecasting is difficult due to low signal-to-noise ratios, latent factors, heavy tails, regime shifts, and jumps.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03184","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_finverbench_8fc73dd6","familyId":"bmf_38fd4857781f","name":"FinVerBench","oneLine":"FinVerBench evaluates financial statement verification in LLMs, using SEC 10-K XBRL filings from 43 S&P 500 companies with a four-category error taxonomy (arithmetic, cross-statement linkage, year-over-year, magnitude perturbations).","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29586","pdf":"https://arxiv.org/pdf/2605.29586","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29586"},"evidence":{"snippet":"We introduce FinVerBench, a benchmark and validity study for financial statement verification: determining whether a set of corporate financial statements is numerically consistent from the information shown to the model.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29586"},"ranking":{},"description":"FinVerBench evaluates financial statement verification in LLMs, using SEC 10-K XBRL filings from 43 S&P 500 companies with a four-category error taxonomy (arithmetic, cross-statement linkage, year-over-year, magnitude perturbations).","whyItMatters":"Financial statement verification requires calibrated judgment under incomplete observability and realistic numerical rendering, not just arithmetic detection. This benchmark provides a reusable diagnostic subset to assess model performance and validity.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9f126b116c4385c947399c391bd3c3e87a8ed679fcb17f6c6fa7631b7a3022dd"},"motivation":"We introduce FinVerBench, a benchmark and validity study for financial statement verification: determining whether a set of corporate financial statements is numerically consistent from the information shown to the model.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29586","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_finverse_651ab4a4","familyId":"bmf_7c67cb28d840","name":"FinVerse","oneLine":"FinVerse is a finance-domain time-series forecasting benchmark with 116,897 series (171.1M observations), selecting 60,232 as evaluation targets. It defines 11 metric families (78 metrics) tailored to each series' economic meaning.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03259","pdf":"https://arxiv.org/pdf/2608.03259","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03259"},"evidence":{"snippet":"To this end, we introduce FinVerse, a finance-domain time-series forecasting benchmark that takes a first step toward more realistic evaluation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03259"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FinVerse is a finance-domain time-series forecasting benchmark with 116,897 series (171.1M observations), selecting 60,232 as evaluation targets. It defines 11 metric families (78 metrics) tailored to each series' economic meaning.","whyItMatters":"Generic forecasting benchmarks use uniform error metrics that may not align with financial decision objectives. This benchmark aims to provide domain-aware evaluation that better reflects real-world utility.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a3fa652ffd2bb2f057ac9fad57ec414ff36f1d19038a8b0af1f073749a2628d7"},"motivation":"As time-series foundation models have emerged, the need for benchmarks that can evaluate their forecasting ability in meaningful ways has become increasingly important.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03259","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_e9a8afcbc786fbeb","familyId":"catalog_family_e9a8afcbc786fbeb","name":"Firefox 147 exploits","oneLine":"Share of patched Firefox 147 JavaScript-engine targets for which the model produced a working arbitrary-code-execution exploit.","description":"Share of patched Firefox 147 JavaScript-engine targets for which the model produced a working arbitrary-code-execution exploit.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.anthropic.com/news/claude-fable-5-mythos-5","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e9a8afcbc786fbeb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/firefox147workingexploit"}],"catalogSources":[{"catalog":"benchlm","sourceId":"firefox147WorkingExploit","url":"https://benchlm.ai/benchmarks/firefox147workingexploit","paperUrl":"https://www.anthropic.com/news/claude-fable-5-mythos-5","year":"2026","fullName":"Firefox 147 Working Exploit Rate","format":"Working arbitrary-code-execution rate","tasks":"250 Firefox 147 vulnerability trials","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_firm-video-bench_74b2abde","familyId":"bmf_42680619c993","name":"FIRM-Video-Bench","oneLine":"FIRM-Video-Bench is a benchmark of 250 text-to-video generations with 750 point-wise human annotations covering Instruction Following, World Coherence, and Perceptual Quality.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-22","firstSeenAt":"2026-08-25","recognitionConfidence":0.65,"links":{"report":"http://arxiv.org/abs/2608.21839v1","pdf":"https://arxiv.org/pdf/2608.21839v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Subsequently, we construct FIRM-Video-90K with 88,044 dimension-specific instances from 29,348 videos, and introduce FIRM-Video-Bench with 750 point-wise human annotations across 250 videos.","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":"2026-08-27T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.21839"},"ranking":{"30d":{"score":42,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"FIRM-Video-Bench is a benchmark of 250 text-to-video generations with 750 point-wise human annotations covering Instruction Following, World Coherence, and Perceptual Quality.","whyItMatters":"It provides a fixed annotated set for evaluating reward models' ability to score videos on fine-grained dimensions, enabling comparison of text-to-video evaluation approaches.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:38:30.498900Z","inputHash":"c3567366b3b4b292c0b491576fc1c7b91b973339abdee8aa94ac64a2299036e6"},"motivation":"Reliable reward models are essential for text-to-video evaluation and alignment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T17:38:30.498900Z","model":"deepseek-v4-pro","decisionReason":"The abstract formally names FIRM-Video-Bench and specifies a public dataset with human annotations, satisfying a reusable evaluation object and scoring contract.","canonicalNameSource":"abstract","canonicalNameEvidence":"introduce FIRM-Video-Bench with 750 point-wise human annotations across 250 videos"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.21839v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":15,"confidence":"Low","horizon":"7d","reason":"A narrowly scoped text-to-video reward modeling benchmark with a moderate dataset size and no independent adoption signals is likely to draw limited immediate attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_fitaqa_e0a97bca","familyId":"bmf_3b8db6654700","name":"FitAQA","oneLine":"FitAQA evaluates fitness action quality assessment in MLLMs across perception, judgement, and temporal grounding tasks, using a unified taxonomy of 38 form errors in six dimensions over 30 exercises.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.08736","pdf":"https://arxiv.org/pdf/2608.08736","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.08736"},"evidence":{"snippet":"We introduce FitAQA, a systematic benchmark for evaluating MLLMs in fitness AQA, containing 2,219 videos and 5,512 QA instances across 30 bodyweight exercises.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08736"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FitAQA evaluates fitness action quality assessment in MLLMs across perception, judgement, and temporal grounding tasks, using a unified taxonomy of 38 form errors in six dimensions over 30 exercises.","whyItMatters":"It provides a systematic benchmark for fitness AQA, enabling assessment of MLLMs' ability to perceive and reason about exercise quality, and identifies visual perception as a key bottleneck.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"21de7b802bd8ae885b3add2d9aa4da55c9bd3bef1b3c613a3199e2b920b327ad"},"motivation":"Fitness Action Quality Assessment (AQA) is important for intelligent sports training, yet the capabilities of Multimodal Large Language Models (MLLMs) in this setting remain underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.08736","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_52d35bcd0e93306f","familyId":"catalog_family_52d35bcd0e93306f","name":"Flame-VLM-Code","oneLine":"Flame-VLM-Code evaluates multimodal models on visual code generation tasks, measuring ability to generate code from visual inputs such as UI mockups and design specifications.","description":"Flame-VLM-Code evaluates multimodal models on visual code generation tasks, measuring ability to generate code from visual inputs such as UI mockups and design specifications.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Code","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_52d35bcd0e93306f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/flamevlmcode"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/flame-vlm-code"}],"catalogSources":[{"catalog":"benchlm","sourceId":"flameVlmCode","url":"https://benchlm.ai/benchmarks/flamevlmcode","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"Flame-VLM-Code","format":"Vision-language code generation","tasks":"Multimodal coding tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"flame-vlm-code","url":"https://llm-stats.com/benchmarks/flame-vlm-code","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","code","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_flamevqa_fd5c1a64","familyId":"bmf_5e31ad44d17e","name":"FlameVQA","oneLine":"FlameVQA is a multiple-choice visual question answering benchmark for UAV-based wildfire monitoring, built on FLAME 3 with paired RGB and radiometric thermal images. It includes 34 questions per image across six capability groups, covering detection, localization, coverage estimation, cross-modal reasoning, and flight planning. Labels are generated via MLLM assistance, deterministic thermal rules, and human auditing. The dataset and code are open-source.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Safety","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.27128","pdf":"https://arxiv.org/pdf/2606.27128","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.27128"},"evidence":{"snippet":"We introduce FlameVQA, a multiple-choice visual question answering benchmark for UAV-based wildfire intelligence built on FLAME 3, leveraging paired RGB imagery and radiometric thermal TIFFs for temperature-grounded, safety-critical reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.27128"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FlameVQA is a multiple-choice visual question answering benchmark for UAV-based wildfire monitoring, built on FLAME 3 with paired RGB and radiometric thermal images. It includes 34 questions per image across six capability groups, covering detection, localization, coverage estimation, cross-modal reasoning, and flight planning. Labels are generated via MLLM assistance, deterministic thermal rules, and human auditing. The dataset and code are open-source.","whyItMatters":"FlameVQA addresses the lack of benchmarks for evaluating vision-language models in safety-critical wildfire scenarios where RGB-only interpretation is insufficient. It provides a standardized evaluation for capabilities like smoke detection and coverage estimation, which are critical for practical deployment of MLLMs in disaster monitoring.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a632b6de7e18eb2936658e9aa29a3bd19fe205b989d42a1ee97d624c48417fea"},"motivation":"Wildfire monitoring from UAVs requires reliable reasoning over complex aerial scenes, where smoke, scale variation, and occlusions often limit RGB-only interpretation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.27128","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_flatlab_52af74dd","familyId":"bmf_694513e572ba","name":"FlatLab","oneLine":"FlatLab is a simulation-based benchmark for robotic manipulation of flat objects, providing physical simulation, data collection, and standardized task definitions and evaluation protocols.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.14049","pdf":"https://arxiv.org/pdf/2608.14049","project":"https://flatlab-web.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.14049"},"evidence":{"snippet":"To enable systematic evaluation, we introduce FlatLab, a comprehensive simulation benchmark for robotic flat object manipulation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14049"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FlatLab is a simulation-based benchmark for robotic manipulation of flat objects, providing physical simulation, data collection, and standardized task definitions and evaluation protocols.","whyItMatters":"It addresses the lack of standardized evaluation for flat object manipulation, a challenging domain due to ungraspable configurations and varied materials.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"354beeaa0dd82828229136b5c06de4ac752c63d6acda8a46c70ecca9456ae104"},"motivation":"Robotic manipulation of flat objects is challenging due to the ungraspable configurations and strong variations in object geometry and material.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026","evidence":"This paper is accepted to ICML 2026","evidenceUrl":"https://arxiv.org/abs/2608.14049","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026","reviewStatus":"accepted","decisionRaw":"This paper is accepted to ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.14049","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"This paper is accepted to ICML 2026","level":"author-claim"}]}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_3628c3b003f98722","familyId":"catalog_family_3628c3b003f98722","name":"FlenQA","oneLine":"Flexible Length Question Answering dataset for evaluating the impact of input length on reasoning performance of language models, featuring True/False questions embedded in contexts of varying lengths (250-3000 tokens) across three reasoning tasks: Monotone Relations, People In Rooms, and simplified Ruletaker","description":"Flexible Length Question Answering dataset for evaluating the impact of input length on reasoning performance of language models, featuring True/False questions embedded in contexts of varying lengths (250-3000 tokens) across three reasoning tasks: Monotone Relations, People In Rooms, and simplified Ruletaker","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/flenqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3628c3b003f98722"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/flenqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"flenqa","url":"https://llm-stats.com/benchmarks/flenqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_a23faa7dfe648a78","familyId":"catalog_family_a23faa7dfe648a78","name":"FLEURS","oneLine":"Few-shot Learning Evaluation of Universal Representations of Speech - a parallel speech dataset in 102 languages built on FLoRes-101 with approximately 12 hours of speech supervision per language for tasks including ASR, speech language identification, translation and retrieval. Scores are shown as speech recognition accuracy (1 - word error rate), so higher is better.","description":"Few-shot Learning Evaluation of Universal Representations of Speech - a parallel speech dataset in 102 languages built on FLoRes-101 with approximately 12 hours of speech supervision per language for tasks including ASR, speech language identification, translation and retrieval. Scores are shown as speech recognition accuracy (1 - word error rate), so higher is better.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Speech To Text"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/fleurs","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a23faa7dfe648a78"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/fleurs"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"fleurs","url":"https://llm-stats.com/benchmarks/fleurs","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","speech to text"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_floatbench_4e9fbef2","familyId":"bmf_04b77197562e","name":"FLOATBench","oneLine":"FLOATBench is a tabular benchmark for surrogate modeling of tower fatigue in 22 MW floating offshore wind turbines. It provides 582,120 per-section fatigue damage labels from OpenFAST simulations across three tower geometries, with regime-aware splits and three evaluation protocols (random, within-tower regime-aware, cross-tower transfer).","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.25717","pdf":"https://arxiv.org/pdf/2605.25717","project":null,"code":"https://github.com/Joao97ribeiro/FLOATBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.25717"},"evidence":{"snippet":"We present FLOATBench, a public tabular benchmark with $582{,}120$ per-section fatigue-damage labels across three $22$ MW FOWT tower geometries, derived from $19{,}404$ high-fidelity OpenFAST simulations across the three towers ($6{,}468$ per tower: $1{,}078$ aligned wind/wave operating points $\\times$ six turbulence seeds), labeled at $30$ cross-sections per tower.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25717"},"ranking":{},"description":"FLOATBench is a tabular benchmark for surrogate modeling of tower fatigue in 22 MW floating offshore wind turbines. It provides 582,120 per-section fatigue damage labels from OpenFAST simulations across three tower geometries, with regime-aware splits and three evaluation protocols (random, within-tower regime-aware, cross-tower transfer).","whyItMatters":"Previously, no shared benchmark existed for FOWT fatigue surrogates, hindering comparison. FLOATBench provides a standardized dataset and protocol, enabling fair evaluation and highlighting regime-aware performance differences.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8c08a697b2f308defb6d96675dfff474a0ae6be9f4233f829379ba72711e5bc0"},"motivation":"Most of the world's offshore wind resource lies in waters too deep for fixed-bottom foundations, making floating offshore wind turbines (FOWTs) essential for deep-water deployment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25717","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DeCoDELab","organizationType":"academic-lab","sourceUrl":"https://github.com/Joao97ribeiro/FLOATBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_floodreasonbench_bd08edc8","familyId":"bmf_c65f287d8be1","name":"FloodReasonBench","oneLine":"FloodReasonBench is a benchmark for vision-language model reasoning segmentation in flood response scenarios. It introduces FloodResponseSeg, a flood-specific dataset, and evaluates pipelines under lightweight visual encoding, split inference, and compressed representations. It also measures accuracy, latency, energy, and communication tradeoffs on an embedded platform.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15410","pdf":"https://arxiv.org/pdf/2608.15410","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.15410"},"evidence":{"snippet":"We present FloodReasonBench, a benchmark for VLM reasoning segmentation for embodied flood response at the edge.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15410"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FloodReasonBench is a benchmark for vision-language model reasoning segmentation in flood response scenarios. It introduces FloodResponseSeg, a flood-specific dataset, and evaluates pipelines under lightweight visual encoding, split inference, and compressed representations. It also measures accuracy, latency, energy, and communication tradeoffs on an embedded platform.","whyItMatters":"Reasoning segmentation for flood response has domain-specific constraints, and this benchmark characterizes model performance and system-level tradeoffs at the edge, which could inform deployment decisions for resource-constrained platforms.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b194550bd74d53b69aa27092dc5f9186a08edb06b78004764fcfa55eda96ab14"},"motivation":"Reasoning segmentation enables vision-language models (VLMs) to translate mission-relevant language requests into pixel-level visual grounding, offering a natural perception interface for embodied agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15410","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_907f245ea307a17b","familyId":"catalog_family_907f245ea307a17b","name":"FLTEval","oneLine":"A repository-level Lean 4 proof engineering benchmark that measures whether a model can complete formal proofs and correctly define new mathematical concepts inside realistic FLT project pull requests.","description":"A repository-level Lean 4 proof engineering benchmark that measures whether a model can complete formal proofs and correctly define new mathematical concepts inside realistic FLT project pull requests.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://mistral.ai/news/leanstral","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_907f245ea307a17b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/flteval"}],"catalogSources":[{"catalog":"benchlm","sourceId":"flteval","url":"https://benchlm.ai/benchmarks/flteval","paperUrl":"https://mistral.ai/news/leanstral","year":"2026","fullName":"FLTEval","format":"Lean 4 repository task completion","tasks":"FLT project pull requests","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_fm-bench_85cc4c12","familyId":"bmf_9ecdad9645f6","name":"FM-Bench","oneLine":"FM-Bench evaluates LLM agents on long-horizon football club management through 20 in-game years with 26 tools, measuring managerial decision quality via a deterministic engine scoring.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-19","firstSeenAt":"2026-08-20","recognitionConfidence":0.75,"links":{"report":"https://arxiv.org/abs/2608.18423","pdf":"https://arxiv.org/pdf/2608.18423","project":null,"code":"https://github.com/Analogy-AI/fm-bench","data":null,"hfPaper":null},"evidence":{"snippet":"FM-Bench (Football Management Benchmark) measures this.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":20,"hfDailySubmittedAt":"2026-08-20T00:00:00.000Z","githubStars":19,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.18423"},"ranking":{"30d":{"score":48,"rank":37,"coverage":0.85,"confidence":"High"},"90d":{"score":46,"rank":104,"coverage":0.7,"confidence":"Medium"}},"description":"FM-Bench evaluates LLM agents on long-horizon football club management through 20 in-game years with 26 tools, measuring managerial decision quality via a deterministic engine scoring.","whyItMatters":"Current agent benchmarks focus on bounded tasks, leaving long-horizon decision-making unmeasured. FM-Bench provides a reproducible environment with deterministic scoring to test sustained decision quality over hundreds of steps.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6e16420e979f7ab35f80e1fb00394624134919f5a8457b2c8552cb1dfa8686c2"},"motivation":"Language model agents now execute bounded tasks reliably.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-20T09:45:58.504563Z","model":"deepseek-v4-flash","decisionReason":"Explicitly designed as a benchmark with clear scoring, code availability, and instructions for running evaluations. The abstract and code repo confirm a public reuse path and ongoing scoring via Arena."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.18423","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-20T09:44:47.308583Z"},"evaluationMode":"score_submission","publishers":[{"name":"Analogy AI","organizationType":"company-research-lab","sourceUrl":"https://github.com/Analogy-AI/fm-bench","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_fmg-bench_a89fdc7a","familyId":"bmf_9aef90db69a1","name":"FMG-Bench","oneLine":"FMG-Bench evaluates English-language Christian theological triage and pastoral guidance in LLMs. It scores whether model responses match expected behaviors for triage levels: primary doctrine, secondary doctrine, prudential, and pastoral, with safety-related escalation appropriate to each scenario. The corpus includes 120 base scenarios plus 37 perturbation variants.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CY"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.12324","pdf":"https://arxiv.org/pdf/2608.12324","project":null,"code":"https://github.com/FideAI/fmg-bench","data":"https://huggingface.co/datasets/FideAI/fmg-bench","hfPaper":"https://huggingface.co/papers/2608.12324"},"evidence":{"snippet":"We introduce FMG-Bench, the Faith & Moral Guidance Benchmark, a 120-scenario benchmark for evaluating large language model behavior in English-language Christian theological triage and pastoral guidance contexts.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":22,"hfDatasetLikes":1},"source":{"type":"arxiv","id":"2608.12324"},"ranking":{},"description":"FMG-Bench evaluates English-language Christian theological triage and pastoral guidance in LLMs. It scores whether model responses match expected behaviors for triage levels: primary doctrine, secondary doctrine, prudential, and pastoral, with safety-related escalation appropriate to each scenario. The corpus includes 120 base scenarios plus 37 perturbation variants.","whyItMatters":"Evaluates a niche but real usage area where models are asked for faith and care advice. Provides a structured scoring contract for safety-critical escalation and robustness to rephrasing, which are not covered by generic QA or safety benchmarks. Useful for developers testing models in contexts where human referral matters.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f053736d5b07605f5f849265e4b768d181e9b5d7d243468d4f37f7021760acdb"},"motivation":"People increasingly ask large language models (LLMs) for counsel on questions of faith, doctrine, and pastoral care.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12324","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_foodmonitor_339fb661","familyId":"bmf_bf993ac0f8c6","name":"FoodMonitor","oneLine":"Evaluates multimodal large language models on explainable compliance analysis in commercial kitchen surveillance, using video clips with dual-channel annotations and a composite metric.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24503","pdf":"https://arxiv.org/pdf/2605.24503","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24503"},"evidence":{"snippet":"We introduce FoodMonitor, a benchmark for explainable compliance analysis in commercial kitchen surveillance.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24503"},"ranking":{},"description":"Evaluates multimodal large language models on explainable compliance analysis in commercial kitchen surveillance, using video clips with dual-channel annotations and a composite metric.","whyItMatters":"Provides a protocol for assessing both spatial localization and semantic rule understanding in video anomaly detection, addressing a gap in existing event-level binary classification benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f0e97ddd21d51305ecaccbde1816523e789e40ab0cfb933f1004921b1d6f56c0"},"motivation":"As AI-powered compliance monitoring becomes increasingly important in public governance and industrial safety, the ability to provide verifiable evidence and traceable accountability signals is essential.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24503","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_force-bench_be1ccc66","familyId":"bmf_3be86d47de12","name":"FORCE-Bench","oneLine":"FORCE-Bench evaluates agentic AI systems in enterprise finance across three task types: financial obligation research, financial entity performance research, and business brief generation. It includes 251 expert-annotated queries and a rubric-based scoring framework across eight dimensions: accuracy, citations, clarity, depth, groundedness, recency, relevance, and structure.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19409","pdf":"https://arxiv.org/pdf/2607.19409","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19409"},"evidence":{"snippet":"We introduce FORCE-Bench, which contains 251 expert-annotated queries and evaluates responses using a rubric-based framework calibrated to the requirements of the operational finance domain, across eight dimensions: accuracy, citations, clarity, depth, groundedness, recency, relevance, and structure.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19409"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FORCE-Bench evaluates agentic AI systems in enterprise finance across three task types: financial obligation research, financial entity performance research, and business brief generation. It includes 251 expert-annotated queries and a rubric-based scoring framework across eight dimensions: accuracy, citations, clarity, depth, groundedness, recency, relevance, and structure.","whyItMatters":"Existing benchmarks focus on general capabilities rather than operational finance workflows. FORCE-Bench provides a domain-specific evaluation tool that measures rule adherence, verifiability, and groundedness, which are critical for real-world deployment of agentic systems in finance. It enables comparable assessment of agent performance under operational constraints.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"23431856bd791d1004934d5b3e7a259a952219e0989a2d3e6faf7295842eb1a5"},"motivation":"Recent advances in large language models have accelerated deployment of agentic systems in operational finance.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19409","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FORCE-Bench team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.19409","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_forcebench_de197b73","familyId":"bmf_3be86d47de12","name":"FORCEBENCH","oneLine":"FORCEBENCH is a contrastive stress test for evidence-force calibration in cited RAG. It pairs fixed cited passages with evidence-calibrated claims and force-raised variants across five axes: relation, modality, scope, temporal validity, and numeric specificity. Evaluation is a fixed, locality-filtered set of 198 pairs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.28044","pdf":"https://arxiv.org/pdf/2605.28044","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28044"},"evidence":{"snippet":"We introduce FORCEBENCH, a contrastive stress test for evidence-force calibration.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28044"},"ranking":{},"description":"FORCEBENCH is a contrastive stress test for evidence-force calibration in cited RAG. It pairs fixed cited passages with evidence-calibrated claims and force-raised variants across five axes: relation, modality, scope, temporal validity, and numeric specificity. Evaluation is a fixed, locality-filtered set of 198 pairs.","whyItMatters":"Current cited RAG evaluation often treats topical relevance as sufficient grounding, missing cases where a relevant source under-warrants an over-strong claim. FORCEBENCH exposes this citation laundering failure and measures evaluator calibration via monotonicity violation rate and force sensitivity.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ee77919ebc271f871f95c08b31ee3200e7432ef59af1bf20dd9ed5cf8c43053e"},"motivation":"Cited RAG evaluation often treats visible sources as a grounding signal, but a real, topically relevant citation can still under-warrant the attached wording.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28044","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_forecastbench-sim_51b8b6fa","familyId":"bmf_6ae3da819bfa","name":"ForecastBench-Sim","oneLine":"A simulated-world forecasting benchmark built on Freeciv game rollouts. Evaluates probabilistic reasoning by scoring forecasts about hidden future game states, with continuous or binary questions, paired intervention worlds, and artifacts for release.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18686","pdf":"https://arxiv.org/pdf/2606.18686","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18686"},"evidence":{"snippet":"We introduce ForecastBench-Sim, a simulated-world forecasting benchmark built on game rollouts from Freeciv, a turn-based strategy game modelled on the Civilization series.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18686"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A simulated-world forecasting benchmark built on Freeciv game rollouts. Evaluates probabilistic reasoning by scoring forecasts about hidden future game states, with continuous or binary questions, paired intervention worlds, and artifacts for release.","whyItMatters":"Addresses the evaluation gap of slow real-world resolution and rare tail events by providing controllable, immediately resolvable forecasting tasks, enabling rigorous study of AI probabilistic reasoning under dynamic states.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"13c680ad7fb6de6d31f24bf5a8af02ae0f341dbae3844150f458bfd8712ba6f5"},"motivation":"Forecasting benchmarks for general-purpose AI systems usually inherit the constraints of the real world: outcomes resolve slowly, tail events are rare, and counterfactual questions are difficult to score.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18686","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_foresci_2aecb7a4","familyId":"bmf_17a312d27c7d","name":"ForeSci","oneLine":"Temporally controlled benchmark with 500 tasks across AI domains for forward-looking research judgment; includes offline knowledge bases and validation protocols.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00644","pdf":"https://arxiv.org/pdf/2606.00644","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.00644"},"evidence":{"snippet":"We introduce ForeSci, a temporally controlled benchmark for evaluating whether LLM agents can make such forward-looking research judgements from historical evidence.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00644"},"ranking":{},"description":"Temporally controlled benchmark with 500 tasks across AI domains for forward-looking research judgment; includes offline knowledge bases and validation protocols.","whyItMatters":"Could support evaluation of research agents in forecasting tasks, but current evidence is insufficient to establish credibility and public accessibility.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dd070da445e2eb6887ed8ed8acf0a8838f626f18df8528670d431868d91e9ddb"},"motivation":"AI research often requires decisions before future evidence exists: which bottleneck to attack, which direction to pursue, or where a project should be positioned.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00644","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_foresightsafety-vla_a9025fe7","familyId":"bmf_f09eeaf1bfcf","name":"ForesightSafety-VLA","oneLine":"A diagnostic safety benchmark for vision-language-action models, evaluating policies across physical interaction, instruction, and perception safety with cumulative cost and risk exposure metrics.","area":"Safety & Trustworthiness","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Multimodal","Safety"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.27079","pdf":"https://arxiv.org/pdf/2606.27079","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.27079"},"evidence":{"snippet":"To address this gap, we introduce ForesightSafety-VLA, a diagnostic benchmark that makes safety the primary evaluation target for VLA systems.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.27079"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A diagnostic safety benchmark for vision-language-action models, evaluating policies across physical interaction, instruction, and perception safety with cumulative cost and risk exposure metrics.","whyItMatters":"The benchmark addresses the lack of systematic safety evaluation for embodied VLA models, offering process-level risk metrics and a taxonomy to localize failure sources, which supports model selection and safety improvements.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"91363d178a06d8cad577c721e4b94b919b1f532e69b5384ec53f38bacda3980f"},"motivation":"In embodied intelligence, safety is a prerequisite for reliable robot deployment in the physical world.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.27079","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_forgetbench_6c7f4d25","familyId":"bmf_6b9c23bcfd72","name":"ForgetBench","oneLine":"Proposes a benchmark for evaluating forgetting dynamics in language models under continual knowledge editing, with concept-based and scenario-based QA, but no public artifacts are available.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.26455","pdf":"https://arxiv.org/pdf/2607.26455","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.26455"},"evidence":{"snippet":"In this work, we propose ForgetBench, a benchmark designed to systematically characterize forgetting behavior in LLMs under continual knowledge editing.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26455"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Proposes a benchmark for evaluating forgetting dynamics in language models under continual knowledge editing, with concept-based and scenario-based QA, but no public artifacts are available.","whyItMatters":"Addresses knowledge retention over time, an important aspect for model updates.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"04aed9ccdd90b360d0698fc8f20e9c70b5f05db1bd14bffa807eee6cb317f187"},"motivation":"Large language models (LLMs) have demonstrated strong capabilities in knowledge acquisition and reasoning, yet their ability to retain previously acquired knowledge under repeated updates remains insufficiently understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26455","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_formaltcs-benchmarking-end-to-end-frontier_6a36f0bb","familyId":"bmf_26d0fc610574","name":"FormalTCS","oneLine":"175 expert-validated instances from STOC, FOCS, SODA, and COLT papers (2025-2026) for evaluating LLMs on end-to-end theoretical computer science research, including autoformalization and proof tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.20153","pdf":"https://arxiv.org/pdf/2608.20153","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce \\ourbenchmark, an expert-validated benchmark for evaluating LLMs on frontier, end-to-end TCS research.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.20153"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"175 expert-validated instances from STOC, FOCS, SODA, and COLT papers (2025-2026) for evaluating LLMs on end-to-end theoretical computer science research, including autoformalization and proof tasks.","whyItMatters":"Offers a realistic research evaluation for LLMs in theoretical computer science, highlighting bottlenecks like autoformalization and research taste.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"b35c0d25f67b7596422b02cb7e655b3c65ccde3f437bd809e2ccb54a80d0502c"},"motivation":"Large language models (LLMs) have shown growing potential for automated theoretical computer science (TCS) research, yet existing benchmarks remain far from realistic research settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The paper explicitly defines the benchmark with expert-validated data and clear metrics, and the abstract indicates public availability of the dataset.","canonicalNameSource":"paper_title","canonicalNameEvidence":"FormalTCS: Benchmarking End-to-End Frontier Formal Theoretical Computer Science Research of Large Language Models"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.20153","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:50.392548Z"},"attentionForecast":{"score":45,"confidence":"Low","horizon":"7d","reason":"The benchmark targets a niche intersection of LLMs and theoretical CS, likely to attract moderate interest from researchers in formal reasoning and theorem proving."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_formstruct-bench_6e003e69","familyId":"bmf_b049023d4081","name":"FormStruct-Bench","oneLine":"The paper introduces FormStruct-Bench, an evaluation dataset for table-form document structure recognition, but no official links to the benchmark artifacts are provided.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.10396","pdf":"https://arxiv.org/pdf/2608.10396","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.10396"},"evidence":{"snippet":"We introduce FormStruct-Bench, a hierarchical and diagnostic benchmark that evaluates table-form document structure recognition at both the document level and progressively finer component levels, allowing aggregate performance to be traced back to specific structural failure modes.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10396"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"The paper introduces FormStruct-Bench, an evaluation dataset for table-form document structure recognition, but no official links to the benchmark artifacts are provided.","whyItMatters":"An evaluation gap exists between holistic document outputs and fine-grained structural scores, but the lack of public benchmark access limits its practical utility.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"03bdec5f287b3a644b8c4bedb02000fe36036d6d5c753cc269d2c2615317d78b"},"motivation":"Transforming table-form documents into machine-processable records requires recovering not only their visible content but also the multilevel structure that organizes it.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10396","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_594cfd2607d4f09e","familyId":"catalog_family_594cfd2607d4f09e","name":"FRAMES","oneLine":"Factuality, Retrieval, And reasoning MEasurement Set - a unified evaluation dataset of 824 challenging multi-hop questions for testing retrieval-augmented generation systems across factuality, retrieval accuracy, and reasoning capabilities, requiring integration of 2-15 Wikipedia articles per question","description":"Factuality, Retrieval, And reasoning MEasurement Set - a unified evaluation dataset of 824 challenging multi-hop questions for testing retrieval-augmented generation systems across factuality, retrieval accuracy, and reasoning capabilities, requiring integration of 2-15 Wikipedia articles per question","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Search"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/frames","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_594cfd2607d4f09e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frames"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"frames","url":"https://llm-stats.com/benchmarks/frames","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","search"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_fraudbench_81b0f089","familyId":"bmf_f628f0268b64","name":"FraudBench","oneLine":"A protocol-sensitive benchmark for adversarial robustness in financial fraud and credit risk, evaluating the same model-attack-defence setting under three protocols: unconstrained, post-hoc filtering, and deployment-aware constraint-integrated attacks.","area":"Safety & Trustworthiness","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":["Robustness"],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.24551","pdf":"https://arxiv.org/pdf/2608.24551","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"This paper presents FraudBench, a protocol-sensitive benchmark for adversarial robustness evaluation in financial fraud and credit-risk detection.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24551"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A protocol-sensitive benchmark for adversarial robustness in financial fraud and credit risk, evaluating the same model-attack-defence setting under three protocols: unconstrained, post-hoc filtering, and deployment-aware constraint-integrated attacks.","whyItMatters":"Highlights that robustness conclusions are protocol-dependent, urging joint reporting of predictive degradation and attack feasibility, and incorporation of domain constraints into attack generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"8e8c5e664cea635997a994aa0543f14c0cd74edb8b20a260edb912a037135eb6"},"motivation":"Machine learning models are widely used in financial fraud and credit-risk detection, yet their adversarial robustness remains difficult to evaluate because financial tabular data involve domain-specific constraints, severe class imbalance, and asymmetric attacker capability.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:08:47.245811Z","model":"deepseek-v4-pro","decisionReason":"Named benchmark with protocol variation, explicit datasets, and a framework for robustness evaluation under domain constraints.","canonicalNameSource":"paper_title","canonicalNameEvidence":"FraudBench: Protocol-Sensitive Benchmarking of Adversarial Robustness for Financial Risk Assessment"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24551","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":38,"confidence":"Low","horizon":"7d","reason":"Financial AI robustness is a timely topic, but lack of public code and data limits immediate engagement."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_freightbidbench_ee372bca","familyId":"bmf_a92125a5e129","name":"FreightBidBench","oneLine":"FreightBidBench is a public-calibrated, closed-loop benchmark for real-time truckload bid acceptance with explicit operational feasibility and economics, including pickup reach, appointment windows, hours-of-service, and yard delays. It provides two full-horizon hindsight ceilings for evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.07343","pdf":"https://arxiv.org/pdf/2607.07343","project":null,"code":"https://github.com/aswincsekar/freightbidbench","data":null,"hfPaper":"https://huggingface.co/papers/2607.07343"},"evidence":{"snippet":"We introduce FreightBidBench, a public-calibrated, dependency-free, closed-loop benchmark in which feasibility (pickup reach, appointment windows, simplified hours-of-service, stochastic yard delays) and economics (service-failure penalty, terminal fleet value, daily price-premium window) are explicit, versioned, and reproducible from public Freight Analysis Framework and U.S.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.07343"},"ranking":{"90d":{"score":23,"rank":374,"coverage":0.55,"confidence":"Low"}},"description":"FreightBidBench is a public-calibrated, closed-loop benchmark for real-time truckload bid acceptance with explicit operational feasibility and economics, including pickup reach, appointment windows, hours-of-service, and yard delays. It provides two full-horizon hindsight ceilings for evaluation.","whyItMatters":"Offers a reproducible benchmark for a dynamic stochastic decision problem that lacked public options, enabling evaluation of bid acceptance policies under operational constraints.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"80bbb71340db600bbf218cb10a7f2743c2adc67f4834693e9a0dd3ebb760ec3e"},"motivation":"Online truckload bid acceptance is a closed-loop stochastic decision problem in which a carrier or broker must, in real time, accept or reject a tendered load subject to operational feasibility, fleet repositioning costs, and opportunity cost against future demand.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.07343","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Aswin C. Sekar","organizationType":"academic-lab","sourceUrl":"https://github.com/aswincsekar/freightbidbench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_1fd0ed42ac0061cc","familyId":"catalog_family_1fd0ed42ac0061cc","name":"French MMLU","oneLine":"French version of MMLU-Pro, a multilingual benchmark for evaluating language models' cross-lingual reasoning capabilities across 14 diverse domains including mathematics, physics, chemistry, law, engineering, psychology, and health.","description":"French version of MMLU-Pro, a multilingual benchmark for evaluating language models' cross-lingual reasoning capabilities across 14 diverse domains including mathematics, physics, chemistry, law, engineering, psychology, and health.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Language","Legal","Reasoning","Finance","General","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/french-mmlu","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1fd0ed42ac0061cc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/french-mmlu"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"french-mmlu","url":"https://llm-stats.com/benchmarks/french-mmlu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","legal","reasoning","finance","general","healthcare"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_freshcache-bench_dd554470","familyId":"bmf_e44b422adf38","name":"FreshCache-Bench","oneLine":"FreshCache-Bench provides 8,072 base queries across five freshness classes with ground truth staleness labels from web snapshots at 1, 12, 24 hours, and 7 days, expanded to 31,201 queries via paraphrase generation, for evaluating semantic caching in retrieval-augmented LLMs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.04281","pdf":"https://arxiv.org/pdf/2607.04281","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.04281"},"evidence":{"snippet":"We introduce FreshCache-Bench, a benchmark of 8,072 base queries across five freshness classes with ground truth staleness labels drawn from real web snapshots at 1, 12, 24 hours, and 7 days after a baseline crawl, expanded to 31,201 queries via paraphrase generation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.04281"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FreshCache-Bench provides 8,072 base queries across five freshness classes with ground truth staleness labels from web snapshots at 1, 12, 24 hours, and 7 days, expanded to 31,201 queries via paraphrase generation, for evaluating semantic caching in retrieval-augmented LLMs.","whyItMatters":"Semantic caching for RAG lacks standardized evaluation of freshness; this benchmark addresses the gap by quantifying stale error and search API savings, enabling comparison of caching strategies in terms of cost and correctness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"682cefc0af17c68f6426c0ac6c4d344b649fe55755c77570df580281e3fa2b5d"},"motivation":"Semantic caching reduces the latency and cost of retrieval-augmented generation (RAG) by serving cached answers to semantically similar queries, but most existing methods do not model the time-varying freshness of open-web evidence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.04281","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_friendbench_e9086fd3","familyId":"bmf_ee7ae5c537f8","name":"FriendBench","oneLine":"FriendBench evaluates dyadic familiarity inference in humans and models using 96 dyads, comparing accuracy across modalities.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.29602","pdf":"https://arxiv.org/pdf/2607.29602","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.29602"},"evidence":{"snippet":"We introduce FriendBench, a benchmark for inferring whether two people are already familiar or are meeting as strangers, from a 20-second clip of a dyadic ice-breaker conversation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.29602"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"FriendBench evaluates dyadic familiarity inference in humans and models using 96 dyads, comparing accuracy across modalities.","whyItMatters":"Provides insight into social perception differences, but does not offer a standardized comparison path for models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"49ae30534c2f22ded8a5e87ad723e2bc4e41fd6cbe1291869577a97c0f11ee9c"},"motivation":"Reading a social situation often depends on behavior, not words alone.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.29602","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_from-financial-sentiment-classification-to_65cf9b8b","familyId":"bmf_322d40bddebf","name":"From Financial Sentiment Classification to Return Predictability","oneLine":"Evaluates LLMs on financial sentiment classification and downstream return predictability using a reproducible QLoRA fine-tuning and evaluation pipeline.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-24","recognitionConfidence":0.4,"links":{"report":"https://github.com/chirindaopensource/from_financial_sentiment_classification_to_return_predictability","pdf":null,"project":"https://img.shields.io/badge/License-MIT-2C3E50?style=flat","code":"https://github.com/chirindaopensource/from_financial_sentiment_classification_to_return_predictability","data":"https://huggingface.co/datasets/financial_phrasebank","hfPaper":null},"evidence":{"snippet":"from_financial_sentiment_classification_to_return_predictability End-to-End Python implementation of Luo's (2026) benchmark construction and evaluation stack for financial NLP.","reasonCodes":["discovered via github","benchmark term in abstract","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:chirindaopensource/from_financial_sentiment_classification_to_return_predictability"},"ranking":{"30d":{"score":23,"rank":123,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":327,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates LLMs on financial sentiment classification and downstream return predictability using a reproducible QLoRA fine-tuning and evaluation pipeline.","whyItMatters":"Connects NLP performance to economic outcomes, offering a reusable stack for assessing whether sentiment accuracy translates into trading signal value.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"ae510a6231f3594c83af0418dc4b1965e85a0e8a6a0d11851bf58d3858aac116"},"motivation":"from_financial_sentiment_classification_to_return_predictability End-to-End Python implementation of Luo's (2026) benchmark construction and evaluation stack for financial NLP.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/chirindaopensource/from_financial_sentiment_classification_to_return_predictability","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-24T07:40:52.396154Z"},"attentionForecast":{"score":44,"confidence":"Low","horizon":"7d","reason":"Financial NLP benchmarking has moderate niche interest, but this standalone implementation lacks a leaderboard or broad model comparison."},"evaluationMode":"public_reusable","publishers":[{"name":"Craig Chirinda (Open Source Projects)","organizationType":"community","sourceUrl":"https://github.com/chirindaopensource/from_financial_sentiment_classification_to_return_predictability","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_frontier-agent-benchmark_f8aadff6","familyId":"bmf_fd79a742d17a","name":"Frontier Agent Benchmark (FAB)","oneLine":"Evaluates autonomous AI engineering agents on engineering quality dimensions such as testing, architecture, and maintainability, using observed telemetry with provenance labels.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/shubhraj5575/frontier-agent-benchmark","pdf":null,"project":"http://localhost:8737","code":"https://github.com/shubhraj5575/frontier-agent-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"frontier-agent-benchmark # Frontier Agent Benchmark (FAB) **Independent observability and benchmarking platform for evaluating autonomous AI engineering agents on engineering quality - not volume.** FAB answers a different question than \"how much code did the agent write?\".","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:shubhraj5575/frontier-agent-benchmark"},"ranking":{"30d":{"score":23,"rank":122,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":326,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates autonomous AI engineering agents on engineering quality dimensions such as testing, architecture, and maintainability, using observed telemetry with provenance labels.","whyItMatters":"Provides a transparent alternative to code-volume metrics by scoring how well agents build working, tested, and maintainable software.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"96c33a40a5d0ae897ea2c4966de66e5e430d0bb8cd30f9692879275fcafb38ea"},"motivation":"frontier-agent-benchmark # Frontier Agent Benchmark (FAB) **Independent observability and benchmarking platform for evaluating autonomous AI engineering agents on engineering quality - not volume.** FAB answers a different question than \"how much code did the agent write?\".","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/shubhraj5575/frontier-agent-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":47,"confidence":"Low","horizon":"7d","reason":"The benchmark addresses a growing need for AI agent evaluation but is currently a single-repository project without demonstrated external submissions."},"evaluationMode":"score_submission","publishers":[{"name":"Frontier Agent Benchmark authors","organizationType":"academic-lab","sourceUrl":"https://github.com/shubhraj5575/frontier-agent-benchmark","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_frontier-financial-judgement_bf94dbc0","familyId":"bmf_3ef91cbe5c1c","name":"Frontier Financial Judgement","oneLine":"Frontier Financial Judgement assesses agents' ability to identify valuation-relevant financial information from 656 synthetic and real news items, matching expert labels.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20645","pdf":"https://arxiv.org/pdf/2607.20645","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20645"},"evidence":{"snippet":"We introduce Frontier Financial Judgement, a challenging new benchmark developed in collaboration with professional equity analysts to assess agents' ability to replicate expert human judgements.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20645"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Frontier Financial Judgement assesses agents' ability to identify valuation-relevant financial information from 656 synthetic and real news items, matching expert labels.","whyItMatters":"News-flow filtering is critical for equity analysts; the benchmark measures agent accuracy, cost, and false positives, but lacks public artifacts for reuse.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"df990acbabc7c6d58d3678a9cd84cf446dbc62cf331ae56dde739e4d0d2837e2"},"motivation":"We introduce Frontier Financial Judgement, a challenging new benchmark developed in collaboration with professional equity analysts to assess agents' ability to replicate expert human judgements.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20645","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_eb184e86d3bc200b","familyId":"catalog_family_eb184e86d3bc200b","name":"Frontier-Bench v0.1","oneLine":"Frontier-Bench v0.1 evaluates agentic terminal coding. Anthropic reports results using the mini-SWE-agent harness and a GKE backend, measured as mean reward across five attempts per task.","description":"Frontier-Bench v0.1 evaluates agentic terminal coding. Anthropic reports results using the mini-SWE-agent harness and a GKE backend, measured as mean reward across five attempts per task.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/frontier-bench-v0.1","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_eb184e86d3bc200b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frontier-bench-v0.1"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"frontier-bench-v0.1","url":"https://llm-stats.com/benchmarks/frontier-bench-v0.1","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_frontierchallenge_5fbc977d","familyId":"bmf_407bf4a4eea5","name":"FrontierChallenge","oneLine":"Evaluates scientific agents on 97 released end-to-end workflows across six domains, using pass rate and average score to measure full delivery of required scientific deliverables.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-27","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.24979","pdf":"https://arxiv.org/pdf/2608.24979","project":"https://apodexai.github.io/FrontierAgent/benchmarks/FrontierChallenge/","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce FrontierChallenge, a cross-domain benchmark comprising 300 end-to-end scientific workflows.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":142,"hfDailySubmittedAt":"2026-08-27T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24979"},"ranking":{"30d":{"score":61,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":56,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates scientific agents on 97 released end-to-end workflows across six domains, using pass rate and average score to measure full delivery of required scientific deliverables.","whyItMatters":"Measures complete workflow execution rather than isolated task success, exposing overclaims by agents and guiding development of more reliable scientific automation.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"097fcc1bb07f60c1afa020e10e9880d101882cb3cf3a6ebf87e54887e8d298b0"},"motivation":"Scientific agents increasingly analyze data, execute code, and produce research artifacts, yet most benchmarks emphasize final answers, isolated programs, or a single domain.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"Defines a clear benchmark with fixed tasks and scoring metrics, and provides a project website; released tasks are available for reuse by other teams.","canonicalNameSource":"paper_title","canonicalNameEvidence":"FrontierChallenge: Evaluating Scientific Workflow Completion"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24979","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":60,"confidence":"Medium","horizon":"7d","reason":"Scientific agent evaluation is an emerging frontier, and the cross-domain workflow completion focus with a dedicated project page may generate early interest."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_99879b83a6d7430a","familyId":"catalog_family_99879b83a6d7430a","name":"FrontierCode","oneLine":"FrontierCode is Cognition's coding evaluation that tests whether models can pass difficult coding tasks while meeting the standards of high-quality production codebases. The Diamond subset contains the hardest problems.","description":"FrontierCode is Cognition's coding evaluation that tests whether models can pass difficult coding tasks while meeting the standards of high-quality production codebases. The Diamond subset contains the hardest problems.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/frontiercode","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_99879b83a6d7430a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frontiercode"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"frontiercode","url":"https://llm-stats.com/benchmarks/frontiercode","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","code"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_dd02ef5d25135618","familyId":"catalog_family_dd02ef5d25135618","name":"FrontierCode 1.1","oneLine":"FrontierCode 1.1 evaluates whether coding-agent changes are mergeable, using unit tests, maintainer-defined rubrics, and verifiers. Runs flagged for unfair internet use receive a zero score.","description":"FrontierCode 1.1 evaluates whether coding-agent changes are mergeable, using unit tests, maintainer-defined rubrics, and verifiers. Runs flagged for unfair internet use receive a zero score.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/frontiercode-1.1","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dd02ef5d25135618"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frontiercode-1.1"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"frontiercode-1.1","url":"https://llm-stats.com/benchmarks/frontiercode-1.1","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","code"],"catalogModelCount":16,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_daaab0754ec79c10","familyId":"catalog_family_daaab0754ec79c10","name":"FrontierCode 1.1 Extended","oneLine":"Cognition's 150-task Extended subset of the FrontierCode 1.1 software-engineering benchmark.","description":"Cognition's 150-task Extended subset of the FrontierCode 1.1 software-engineering benchmark.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://devin.ai/blog/gpt-5-6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_daaab0754ec79c10"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/frontiercode11extended"}],"catalogSources":[{"catalog":"benchlm","sourceId":"frontierCode11Extended","url":"https://benchlm.ai/benchmarks/frontiercode11extended","paperUrl":"https://devin.ai/blog/gpt-5-6","year":"2026","fullName":"FrontierCode 1.1 Extended","format":"Repository task completion with maintainer rubrics","tasks":"150 private software-engineering tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_57a2ae818f3b1016","familyId":"catalog_family_57a2ae818f3b1016","name":"FrontierCode 1.1 Main","oneLine":"Cognition's 100-task software-engineering benchmark for whether coding agents produce mergeable, production-quality pull requests, scored for correctness, tests, scope, style, and maintainability through maintainer-authored rubrics.","description":"Cognition's 100-task software-engineering benchmark for whether coding agents produce mergeable, production-quality pull requests, scored for correctness, tests, scope, style, and maintainability through maintainer-authored rubrics.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://cognition.com/frontiercode","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_57a2ae818f3b1016"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/frontiercode"}],"catalogSources":[{"catalog":"benchlm","sourceId":"frontierCode","url":"https://benchlm.ai/benchmarks/frontiercode","paperUrl":"https://cognition.com/frontiercode","year":"2026","fullName":"FrontierCode 1.1 Main","format":"Repository task completion with maintainer rubrics","tasks":"100 private Main tasks (150 in Extended)","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_dbd18e51dd070805","familyId":"catalog_family_dbd18e51dd070805","name":"FrontierCS","oneLine":"FrontierCS is a benchmark of frontier computer-science problems requiring deep theoretical understanding and rigorous multi-step reasoning at the edge of the field.","description":"FrontierCS is a benchmark of frontier computer-science problems requiring deep theoretical understanding and rigorous multi-step reasoning at the edge of the field.","area":"Code & Software","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Reasoning","Science","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/frontiercs","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dbd18e51dd070805"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frontiercs"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"frontiercs","url":"https://llm-stats.com/benchmarks/frontiercs","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","science","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"specific"},{"id":"catalog_c563b6543b2f3c47","familyId":"catalog_family_c563b6543b2f3c47","name":"FrontierCyber","oneLine":"Independent evaluation of AI agents against vulnerable real-world systems in dynamic environments.","description":"Independent evaluation of AI agents against vulnerable real-world systems in dynamic environments.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.irregular.com/research/frontiercyber","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c563b6543b2f3c47"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/frontiercyber"}],"catalogSources":[{"catalog":"benchlm","sourceId":"frontierCyber","url":"https://benchlm.ai/benchmarks/frontiercyber","paperUrl":"https://www.irregular.com/research/frontiercyber","year":"2026","fullName":"FrontierCyber","format":"Tasks solved","tasks":"197 dynamic cyber tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_frontierfinance_f2f44751","familyId":"bmf_d3024ba14951","name":"FrontierFinance","oneLine":"Comprises 220 expert-crafted queries with 11,543 source-attributed rubrics across six finance use cases, evaluating agents on open-ended analyst-style answers via rubric-based grading.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.11683","pdf":"https://arxiv.org/pdf/2608.11683","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.11683"},"evidence":{"snippet":"We introduce FrontierFinance, a fully open benchmark of 220 expert-crafted queries and 11,543 source-attributed rubrics spanning six crucial use cases across the full investor workflow.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11683"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Comprises 220 expert-crafted queries with 11,543 source-attributed rubrics across six finance use cases, evaluating agents on open-ended analyst-style answers via rubric-based grading.","whyItMatters":"Captures the full investor workflow beyond narrow data extraction, offering a harder and broader evaluation that assesses agent quality and efficiency in realistic finance research.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"7fd3ddaacac95ccc812e5910cf0938dc4d139d8776cfa1f2e38ca412fda17376"},"motivation":"AI agents are increasingly deployed for professional investment research, yet no benchmark captures the complexity of the full investor workflow.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11683","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Samaya AI","organizationType":"company-research-lab","sourceUrl":"https://arxiv.org/abs/2608.11683","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_2f5bfab07dfbb484","familyId":"catalog_family_2f5bfab07dfbb484","name":"FrontierMath","oneLine":"A benchmark of hundreds of original, exceptionally challenging mathematics problems crafted and vetted by expert mathematicians, covering most major branches of modern mathematics from number theory and real analysis to algebraic geometry and category theory.","description":"A benchmark of hundreds of original, exceptionally challenging mathematics problems crafted and vetted by expert mathematicians, covering most major branches of modern mathematics from number theory and real analysis to algebraic geometry and category theory.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/frontiermath","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2f5bfab07dfbb484"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frontiermath"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"frontiermath","url":"https://llm-stats.com/benchmarks/frontiermath","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":17,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_a25fb18142da2dbf","familyId":"catalog_family_a25fb18142da2dbf","name":"FrontierMath (legacy)","oneLine":"Legacy FrontierMath values retained for historical model pages. This field is not used in current rankings because it can mix prior benchmark versions and slices.","description":"Legacy FrontierMath values retained for historical model pages. This field is not used in current rankings because it can mix prior benchmark versions and slices.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://epoch.ai/frontiermath","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a25fb18142da2dbf"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/frontiermath"}],"catalogSources":[{"catalog":"benchlm","sourceId":"frontierMath","url":"https://benchlm.ai/benchmarks/frontiermath","paperUrl":"https://epoch.ai/frontiermath","year":"2024","fullName":"FrontierMath legacy aggregate","format":"Open-ended mathematical reasoning with tool access","tasks":"Historical aggregate","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_1a70ebfd035b74be","familyId":"catalog_family_1a70ebfd035b74be","name":"FrontierMath Tier 4 (v2)","oneLine":"FrontierMath Tier 4 subset from the v2 evaluation release.","description":"FrontierMath Tier 4 subset from the v2 evaluation release.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/frontiermath-tier-4-v2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1a70ebfd035b74be"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frontiermath-tier-4-v2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"frontiermath-tier-4-v2","url":"https://llm-stats.com/benchmarks/frontiermath-tier-4-v2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_9193b01245c728ca","familyId":"catalog_family_9193b01245c728ca","name":"FrontierMath v2 (Tier 4)","oneLine":"Epoch AI's corrected v2 Tier 4 expansion, a separate set of exceptionally difficult research-level mathematics problems evaluated with Python-enabled iterative reasoning.","description":"Epoch AI's corrected v2 Tier 4 expansion, a separate set of exceptionally difficult research-level mathematics problems evaluated with Python-enabled iterative reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://epoch.ai/benchmarks/frontiermath-tier-4-v2?view=graph&tab=leaderboard","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9193b01245c728ca"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/frontiermathv2tier4"}],"catalogSources":[{"catalog":"benchlm","sourceId":"frontierMathV2Tier4","url":"https://benchlm.ai/benchmarks/frontiermathv2tier4","paperUrl":"https://epoch.ai/benchmarks/frontiermath-tier-4-v2?view=graph&tab=leaderboard","year":"2026","fullName":"FrontierMath v2 Tier 4","format":"Python-enabled iterative mathematical problem solving","tasks":"43 private extreme-difficulty mathematics problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_8b8a2b18f81e3064","familyId":"catalog_family_8b8a2b18f81e3064","name":"FrontierMath v2 (Tiers 1-3)","oneLine":"Epoch AI's corrected v2 core FrontierMath suite of private advanced mathematics problems. Models can reason iteratively and use Python; scores are pass rates on the private set.","description":"Epoch AI's corrected v2 core FrontierMath suite of private advanced mathematics problems. Models can reason iteratively and use Python; scores are pass rates on the private set.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://epoch.ai/benchmarks/frontiermath-tier-4-v2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8b8a2b18f81e3064"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/frontiermathv2tiers13"}],"catalogSources":[{"catalog":"benchlm","sourceId":"frontierMathV2Tiers13","url":"https://benchlm.ai/benchmarks/frontiermathv2tiers13","paperUrl":"https://epoch.ai/benchmarks/frontiermath-tier-4-v2","year":"2026","fullName":"FrontierMath v2 Tiers 1-3","format":"Python-enabled iterative mathematical problem solving","tasks":"295 private advanced mathematics problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_29b51586e16c2654","familyId":"catalog_family_29b51586e16c2654","name":"FrontierScience","oneLine":"Frontier Science is a benchmark of exceptionally challenging scientific reasoning problems spanning advanced natural-science domains, designed to test expert-level scientific understanding and multi-step reasoning.","description":"Frontier Science is a benchmark of exceptionally challenging scientific reasoning problems spanning advanced natural-science domains, designed to test expert-level scientific understanding and multi-step reasoning.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/frontierscience/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_29b51586e16c2654"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/frontierscience"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frontier-science"}],"catalogSources":[{"catalog":"benchlm","sourceId":"frontierScience","url":"https://benchlm.ai/benchmarks/frontierscience","paperUrl":"https://openai.com/index/frontierscience/","year":"2026","fullName":"FrontierScience","format":"Scientific reasoning benchmark","tasks":"Research-level science tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"frontier-science","url":"https://llm-stats.com/benchmarks/frontier-science","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_845200828dcdd749","familyId":"catalog_family_845200828dcdd749","name":"FrontierScience Olympiad","oneLine":"FrontierScience Olympiad is a benchmark of olympiad-level scientific reasoning problems, testing expert understanding and multi-step reasoning across advanced natural-science domains.","description":"FrontierScience Olympiad is a benchmark of olympiad-level scientific reasoning problems, testing expert understanding and multi-step reasoning across advanced natural-science domains.","area":"Mathematical Reasoning","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning","Science"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/frontierscience-olympiad","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_845200828dcdd749"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frontierscience-olympiad"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"frontierscience-olympiad","url":"https://llm-stats.com/benchmarks/frontierscience-olympiad","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning","science"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"catalog_24a4551d7f8a8078","familyId":"catalog_family_24a4551d7f8a8078","name":"FrontierScience Research","oneLine":"FrontierScience Research is a benchmark evaluating AI models on cutting-edge scientific research questions requiring deep domain expertise, multi-step reasoning, and synthesis of complex scientific concepts across disciplines.","description":"FrontierScience Research is a benchmark evaluating AI models on cutting-edge scientific research questions requiring deep domain expertise, multi-step reasoning, and synthesis of complex scientific concepts across disciplines.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","Science"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_24a4551d7f8a8078"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/frontierscienceresearch"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frontierscience-research"}],"catalogSources":[{"catalog":"benchlm","sourceId":"frontierScienceResearch","url":"https://benchlm.ai/benchmarks/frontierscienceresearch","paperUrl":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","year":"2026","fullName":"FrontierScience Research","format":"Research evaluation","tasks":"Scientific research problems","successorKey":null},{"catalog":"llm-stats","sourceId":"frontierscience-research","url":"https://llm-stats.com/benchmarks/frontierscience-research","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","science"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_4d84bdec793147a5","familyId":"catalog_family_4d84bdec793147a5","name":"FrontierSWE","oneLine":"FrontierSWE measures whether an agent can complete open-ended technical projects at the scale of hours to tens of hours, spanning systems optimization, large-scale code construction, and applied ML research. Performance is reported as a dominance score, where higher is better.","description":"FrontierSWE measures whether an agent can complete open-ended technical projects at the scale of hours to tens of hours, spanning systems optimization, large-scale code construction, and applied ML research. Performance is reported as a dominance score, where higher is better.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.frontierswe.com/blog","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4d84bdec793147a5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/frontierswe"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frontierswe"}],"catalogSources":[{"catalog":"benchlm","sourceId":"frontierSwe","url":"https://benchlm.ai/benchmarks/frontierswe","paperUrl":"https://www.frontierswe.com/blog","year":"2026","fullName":"FrontierSWE","format":"Mean@5, best@5, average rank, and dominance","tasks":"17 ultra-long-horizon engineering and research tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"frontierswe","url":"https://llm-stats.com/benchmarks/frontierswe","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","agents","code"],"catalogModelCount":16,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_c74640a9725f90bd","familyId":"catalog_family_c74640a9725f90bd","name":"FrontierSWE (Impl.)","oneLine":"FrontierSWE (Impl.) evaluates software engineering implementation ability and reports model ranking on implementation tasks. Lower rank is better.","description":"FrontierSWE (Impl.) evaluates software engineering implementation ability and reports model ranking on implementation tasks. Lower rank is better.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/frontier-swe-impl","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c74640a9725f90bd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/frontier-swe-impl"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"frontier-swe-impl","url":"https://llm-stats.com/benchmarks/frontier-swe-impl","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_98f45d18494cedc9","familyId":"catalog_family_98f45d18494cedc9","name":"FullStackBench en","oneLine":"English subset of FullStackBench for evaluating end-to-end software engineering and full-stack development capability.","description":"English subset of FullStackBench for evaluating end-to-end software engineering and full-stack development capability.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/fullstackbench-en","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_98f45d18494cedc9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/fullstackbench-en"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"fullstackbench-en","url":"https://llm-stats.com/benchmarks/fullstackbench-en","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","code"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_41a7e9d5dcfe0e8b","familyId":"catalog_family_41a7e9d5dcfe0e8b","name":"FullStackBench zh","oneLine":"Chinese subset of FullStackBench for evaluating end-to-end software engineering and full-stack development capability.","description":"Chinese subset of FullStackBench for evaluating end-to-end software engineering and full-stack development capability.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/fullstackbench-zh","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_41a7e9d5dcfe0e8b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/fullstackbench-zh"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"fullstackbench-zh","url":"https://llm-stats.com/benchmarks/fullstackbench-zh","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","code"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_341e245f91f03a8b","familyId":"catalog_family_341e245f91f03a8b","name":"FunctionalMATH","oneLine":"A functional variant of the MATH benchmark that tests language models' ability to generalize reasoning patterns across different problem instances, revealing the reasoning gap between static and functional performance.","description":"A functional variant of the MATH benchmark that tests language models' ability to generalize reasoning patterns across different problem instances, revealing the reasoning gap between static and functional performance.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/functionalmath","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_341e245f91f03a8b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/functionalmath"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"functionalmath","url":"https://llm-stats.com/benchmarks/functionalmath","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_funpiq_ac17b6fd","familyId":"bmf_d55ffcdf5c6f","name":"FunPiQ","oneLine":"FunPiQ is a benchmark for pixel-level fundus image quality assessment with pixel-level annotations for anatomical visibility, using a three-class scheme (good, usable, bad). It also introduces EFIQA-CP, an explainable method trained with pseudo-labels.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.25915","pdf":"https://arxiv.org/pdf/2606.25915","project":null,"code":"https://github.com/penway/FunPiQ","data":null,"hfPaper":"https://huggingface.co/papers/2606.25915"},"evidence":{"snippet":"In this work, we introduce FunPiQ, the first FIQA benchmark to provide pixel-level quality annotations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.25915"},"ranking":{"90d":{"score":23,"rank":306,"coverage":0.7,"confidence":"Medium"}},"description":"FunPiQ is a benchmark for pixel-level fundus image quality assessment with pixel-level annotations for anatomical visibility, using a three-class scheme (good, usable, bad). It also introduces EFIQA-CP, an explainable method trained with pseudo-labels.","whyItMatters":"Existing FIQA benchmarks provide only image-level labels, limiting localized degradation analysis. FunPiQ enables task-agnostic explainable quality evaluation through pixel-level ground truth and offers a public dataset for method comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cb7daff2c922718533100460e5fa5b1c1c9fe3e8e47e882df05bab32cf186c94"},"motivation":"Color fundus photography (CFP) is the most common ophthalmic imaging modality for large-scale screening.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"MICCAI 2026 main conference","evidence":"Accepted at MICCAI 2026 main conference. Our code, weights, and dataset are available at https://github.com/penway/FunPiQ","evidenceUrl":"https://arxiv.org/abs/2606.25915","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"MICCAI 2026 main conference","reviewStatus":"accepted","decisionRaw":"Accepted at MICCAI 2026 main conference. Our code, weights, and dataset are available at https://github.com/penway/FunPiQ","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.25915","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at MICCAI 2026 main conference. Our code, weights, and dataset are available at https://github.com/penway/FunPiQ","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_fuzzingbrain-bench_87c92c01","familyId":"bmf_9a7e7c8791a6","name":"FuzzingBrain-Bench","oneLine":"Evaluates LLMs on discovering distinct crashes in 77 open-source software challenges using sanitizer-instrumented Docker harnesses and deterministic scoring.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-25","firstSeenAt":"2026-08-27","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.25158","pdf":"https://arxiv.org/pdf/2608.25158","project":null,"code":"https://github.com/fuzzingbrain/FuzzingBrain-Bench","data":null,"hfPaper":null},"evidence":{"snippet":"We present FuzzingBrain-Bench, a benchmark for assessing AI models' ability to discover bugs in open-source software.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25158"},"ranking":{"30d":{"score":36,"rank":59,"coverage":0.55,"confidence":"Low"},"90d":{"score":35,"rank":197,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates LLMs on discovering distinct crashes in 77 open-source software challenges using sanitizer-instrumented Docker harnesses and deterministic scoring.","whyItMatters":"Shifts vulnerability evaluation from predefined targets to open-ended crash discovery, providing a realistic and reproducible measure of LLM bug-finding capability.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"eea675d0fa8bd1335e3f6f0b46e7129fc934826ef91f1a0eae7e29b1fc5b77f0"},"motivation":"Evaluating the ability of large language models (LLMs) to discover software bugs is increasingly important.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"Corpus and harnesses are publicly available on GitHub with deterministic scoring and a leaderboard-style comparison path, qualifying as a formal benchmark.","canonicalNameSource":"paper_title","canonicalNameEvidence":"FuzzingBrain-Bench V1: Evaluating Open-Ended Bug Discovery by LLMs"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25158","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":68,"confidence":"Medium","horizon":"7d","reason":"Software security and LLM agent evaluation are active areas, and the practical open-ended bug discovery benchmark with public code is likely to attract attention from both communities."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_g-idiomalign_39650ae3","familyId":"bmf_eb51932671bc","name":"G-IdiomAlign","oneLine":"G-IdiomAlign is a gloss-pivoted benchmark for cross-lingual idiom alignment, with protocols for multiple-choice equivalence and gloss-contrastive generation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18989","pdf":"https://arxiv.org/pdf/2606.18989","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18989"},"evidence":{"snippet":"We present G-IdiomAlign, a gloss-pivoted benchmark where each idiom is anchored by an English gloss from Wiktionary.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18989"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"G-IdiomAlign is a gloss-pivoted benchmark for cross-lingual idiom alignment, with protocols for multiple-choice equivalence and gloss-contrastive generation.","whyItMatters":"Idiom translation is challenging due to non-compositionality. This benchmark could support evaluation of multilingual models, but its public availability and scoring contract are not fully clear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3ab442b80666c7e46cfee103fb07c7ec581c30864e51b0fac1b83ec948ce2b69"},"motivation":"Idioms are difficult to transfer across languages due to their non-compositionality and weak surface-form grounding, making literal mappings unreliable.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ACL 2026","evidence":"Accepted to ACL 2026","evidenceUrl":"https://arxiv.org/abs/2606.18989","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ACL 2026","reviewStatus":"accepted","decisionRaw":"Accepted to ACL 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.18989","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to ACL 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_gabench_d11355fd","familyId":"bmf_239d1223e7ea","name":"GABench","oneLine":"GABench evaluates LLM agents on graph analysis tasks across three graph types and four task categories: graph retrieval, graph theory, graph machine learning, and graph open-ended QA. It provides 84 executable tools and 10,400 tasks with verifiable ground truth.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.01684","pdf":"https://arxiv.org/pdf/2608.01684","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01684"},"evidence":{"snippet":"To address these limitations, we introduce GABench, a comprehensive benchmark for agentic graph analysis.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01684"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GABench evaluates LLM agents on graph analysis tasks across three graph types and four task categories: graph retrieval, graph theory, graph machine learning, and graph open-ended QA. It provides 84 executable tools and 10,400 tasks with verifiable ground truth.","whyItMatters":"Existing graph benchmarks lack coverage and typically format tasks as text QA, limiting agent evaluation. GABench offers a comprehensive, tool-based benchmark for assessing end-to-end agentic capabilities in graph analysis, providing practical insights into harness and tool-call quality.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4ac59933601519e111a1d238e66e2b720e26a7ffd9d23282afb3e6862415541c"},"motivation":"Large language model (LLM) agents are increasingly capable of planning, using tools, and interacting with external environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.01684","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"lib_gaia","familyId":"family_gaia","name":"GAIA","oneLine":"Established benchmark family · Agents & Tool Use.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents & Tool Use"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2023-01-01","releaseDatePrecision":"year","firstRelease":{"year":2023,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2311.12983","pdf":null,"project":"https://huggingface.co/gaia-benchmark","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_gaia"},"ranking":{},"recordType":"family","aliases":["General AI Assistants benchmark"],"sourceAttribution":[{"role":"benchmark-paper","url":"https://arxiv.org/abs/2311.12983"}],"adoptionRefs":["google-gemini25"],"modelReportReferences":[{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"gaia","url":"https://benchlm.ai/benchmarks/gaia","paperUrl":null,"year":2024,"fullName":"General AI Assistants","format":null,"tasks":466,"successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0},{"id":"catalog_0197a6c5271dada4","familyId":"catalog_family_0197a6c5271dada4","name":"GAIA2","oneLine":"GAIA2 evaluates general-purpose AI agents on real-world, multi-step questions that require reasoning, tool use, and information retrieval.","description":"GAIA2 evaluates general-purpose AI agents on real-world, multi-step questions that require reasoning, tool use, and information retrieval.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","General","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/gaia2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0197a6c5271dada4"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gaia2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"gaia2","url":"https://llm-stats.com/benchmarks/gaia2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","agents","tool calling"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_gama-bench_3cec2065","familyId":"bmf_e443c7d96cbb","name":"GAMA-Bench","oneLine":"GAMA-Bench evaluates LLMs on gender-asymmetric moral framing across 1,298 paired conflict scenarios, measuring response differences in punitive, therapeutic, and blame dimensions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-12","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.14068","pdf":"https://arxiv.org/pdf/2606.14068","project":null,"code":"https://github.com/xufeiqiong/GAMA-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.14068"},"evidence":{"snippet":"We introduce GAMA-Bench, a gender-mirrored benchmark of 1,298 scenarios covering intimate relationship and public social conflicts.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":10,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.14068"},"ranking":{"90d":{"score":40,"rank":145,"coverage":0.55,"confidence":"Low"}},"description":"GAMA-Bench evaluates LLMs on gender-asymmetric moral framing across 1,298 paired conflict scenarios, measuring response differences in punitive, therapeutic, and blame dimensions.","whyItMatters":"detects biases beyond stereotypes by comparing responses to matched male and female actors, informing fairness in AI decision-making.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"61b891d419f1b1cd095b0519941dab27866e3aa7645a14710bdd707f759b7a1b"},"motivation":"Existing studies on gender bias in LLMs have largely focused on stereotypes, occupational associations, or explicit harmful outputs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.14068","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_7ec5036cca836219","familyId":"catalog_family_7ec5036cca836219","name":"GameDevBench","oneLine":"Evaluates coding agents on 333 multimodal game-development tasks in Godot, spanning 2D graphics, 3D graphics, user interfaces, and gameplay logic.","description":"Evaluates coding agents on 333 multimodal game-development tasks in Godot, spanning 2D graphics, 3D graphics, user interfaces, and gameplay logic.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2602.11103","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7ec5036cca836219"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gamedevbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"gameDevBench","url":"https://benchlm.ai/benchmarks/gamedevbench","paperUrl":"https://arxiv.org/abs/2602.11103","year":"2026","fullName":"GameDevBench","format":"Pass@1 on the full task set with 95% confidence intervals","tasks":"333 tasks from 88 tutorials","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_gameenginebench_f2b34812","familyId":"bmf_d20da2cff03c","name":"GameEngineBench","oneLine":"GameEngineBench evaluates coding agents on scoped C++ implementation tasks within Unreal Engine 5 projects, built from nine real-world game repositories. The 110 tasks span gameplay, multiplayer, AI, animation, UI, and other areas, requiring native C++ changes that compile and pass behavioral tests.","area":"Code & Software","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Software & Cloud","Manufacturing"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.03525","pdf":"https://arxiv.org/pdf/2607.03525","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.03525"},"evidence":{"snippet":"We present GameEngineBench, a benchmark for evaluating coding agents on scoped C++ implementation tasks inside Unreal Engine 5 projects, built from nine real-world game repositories.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.03525"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"GameEngineBench evaluates coding agents on scoped C++ implementation tasks within Unreal Engine 5 projects, built from nine real-world game repositories. The 110 tasks span gameplay, multiplayer, AI, animation, UI, and other areas, requiring native C++ changes that compile and pass behavioral tests.","whyItMatters":"Game-engine development presents unique challenges for coding agents, including stateful, real-time, and interactive systems. This benchmark fills a gap by focusing on deeply integrated C++ tasks, revealing limitations of current agents and providing a practical evaluation for real-world software engineering.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"89ae96020cbf0adb37fa48279cc81e3637acd081e7713d9cfcfcf231c06ad610"},"motivation":"Game engines provide real-time simulation, rendering, physics, interaction, networking, and asset pipelines, making them valuable not only for games but also for 3D applications in healthcare, robotics, architecture, manufacturing, and related domains.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.03525","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GameEngineBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.03525","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"specific"},{"id":"catalog_185897bc155c560e","familyId":"catalog_family_185897bc155c560e","name":"GameWorld","oneLine":"GameWorld evaluates agents on interactive game environments, testing perception, planning, and sequential decision-making to accomplish in-game objectives.","description":"GameWorld evaluates agents on interactive game environments, testing perception, planning, and sequential decision-making to accomplish in-game objectives.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/gameworld","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_185897bc155c560e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gameworld"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"gameworld","url":"https://llm-stats.com/benchmarks/gameworld","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_gamexpert-bench_691da8d8","familyId":"bmf_343bc4dacfb9","name":"GameXpert-Bench","oneLine":"Evaluates coding agents across three game development lifecycle tracks: generation, bug diagnosis and repair, and multi-turn optimization, using live interaction, behavioral tests, and product criteria.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-22","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.21833v1","pdf":"https://arxiv.org/pdf/2608.21833v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Therefore, we introduce GameXpert-Bench, which operationalizes the three lifecycle stages as three complementary benchmark tracks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":17,"hfDailySubmittedAt":"2026-08-25T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.21833"},"ranking":{"30d":{"score":50,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":50,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates coding agents across three game development lifecycle tracks: generation, bug diagnosis and repair, and multi-turn optimization, using live interaction, behavioral tests, and product criteria.","whyItMatters":"Provides a comprehensive benchmark for game development with coding agents, measuring both product quality and process capabilities across the full lifecycle.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:38:30.498900Z","inputHash":"100349822a5c8bd9dce7418c55bf7b909c17fbdaa688d4be5e5a5bbdfe006aeb"},"motivation":"Recent large language models (LLMs) can operate as coding agents that build complete games from natural language requests.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T17:38:30.498900Z","model":"deepseek-v4-pro","decisionReason":"GameXpert-Bench is a formally named benchmark with three tracks and evaluation methods; the paper implies release of the suite for other teams.","canonicalNameSource":"paper_title","canonicalNameEvidence":"GameXpert-Bench: How Far Are Coding Agents from Expert Game Development?"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.21833v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":66,"confidence":"Medium","horizon":"7d","reason":"Game development is a novel and engaging application area for coding agents, though the benchmark's size and complexity may limit immediate adoption."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_gatemem_7d143ff2","familyId":"bmf_4f489248c2fa","name":"GateMem","oneLine":"GateMem evaluates memory governance in multi-principal shared-memory agents across medical, office, education, and household domains. It measures utility, access control, and active forgetting via checkpoints and a composite score.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18829","pdf":"https://arxiv.org/pdf/2606.18829","project":null,"code":"https://github.com/rzhub/GateMem","data":"https://huggingface.co/datasets/Ray368/GateMem","hfPaper":"https://huggingface.co/papers/2606.18829"},"evidence":{"snippet":"We introduce GateMem, a benchmark for multi-principal shared-memory agents.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":18,"hfDailySubmittedAt":"2026-06-22T00:00:00.000Z","githubStars":139,"githubScope":"benchmark_repo","hfDatasetDownloads":178,"hfDatasetLikes":4},"source":{"type":"arxiv","id":"2606.18829"},"ranking":{"90d":{"score":58,"rank":26,"coverage":1.0,"confidence":"High","datasetDownloadRank":40,"datasetRankPopulation":66}},"description":"GateMem evaluates memory governance in multi-principal shared-memory agents across medical, office, education, and household domains. It measures utility, access control, and active forgetting via checkpoints and a composite score.","whyItMatters":"Shared-memory agents are understudied, yet crucial for institutional deployments. GateMem addresses the need for evaluating governance capabilities beyond simple recall, informing development of reliable multi-user agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"71def883b11a4a08b801594c7a902ee5fe8c67fd7bb7c1f56eba440b8034bfa0"},"motivation":"Memory benchmarks for LLM agents largely assume single-user settings, leaving shared assistants for hospitals, workplaces, campuses, and households understudied.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18829","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GateMem Project","organizationType":"academic-lab","sourceUrl":"https://github.com/rzhub/GateMem","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_gauge_6ce93fe5","familyId":"bmf_a90fd9a9a1e6","name":"GAUGE","oneLine":"GAUGE evaluates physical fidelity of simulation engines and video world models using 22 task families covering rigid bodies, cables, textiles, and deformable objects, with real-world trajectories and calibrated metadata.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.05948","pdf":"https://arxiv.org/pdf/2608.05948","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.05948"},"evidence":{"snippet":"We introduce GAUGE, a real-world-grounded diagnostic benchmark for jointly evaluating how numerical simulators and generative video world models reproduce or deviate from real-world physics.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.05948"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GAUGE evaluates physical fidelity of simulation engines and video world models using 22 task families covering rigid bodies, cables, textiles, and deformable objects, with real-world trajectories and calibrated metadata.","whyItMatters":"Existing evaluations of physical fidelity rely on perceptual similarity or human judgment. GAUGE provides a diagnostic benchmark with grounded physical measurements, enabling systematic comparison of simulators and world models on specific physical principles.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8af04eebd1bfdb552f8a4266debabf9d8e10444d4131d466edf1cbe21c6e08ec"},"motivation":"Physics engines facilitate large-scale training and evaluation for embodied intelligence, while generative video world models are emerging as implicit simulators of future states and interactions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.05948","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_gauge_c5d48106","familyId":"bmf_a90fd9a9a1e6","name":"GAUGE","oneLine":"GAUGE evaluates agent-built financial valuation models against observed analyst practice using 56 facets, eight validity gates, and a failure-aware score over 196 tasks.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.24889","pdf":"https://arxiv.org/pdf/2607.24889","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.24889"},"evidence":{"snippet":"We introduce GAUGE, a benchmark for evaluating agent-built valuation models against observed analyst practice rather than a single point answer.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24889"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GAUGE evaluates agent-built financial valuation models against observed analyst practice using 56 facets, eight validity gates, and a failure-aware score over 196 tasks.","whyItMatters":"Provides a benchmark that avoids penalizing legitimate disagreement, enabling fair comparison of agents on financial modeling and judgment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"241f8c88b248e236fca5562ceb58f39ba684b023b452289e162279bcbbb90989"},"motivation":"Financial models combine public disclosures with analyst assumptions to produce forecasts and valuations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24889","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_gauntletbench_be5725f6","familyId":"bmf_44f425325bfc","name":"GauntletBench","oneLine":"GauntletBench evaluates agent generalisation across five professional web applications with 100 vision-intensive tasks, probing temporal perception, graphical understanding, and 3D reasoning via automated objective scoring.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Geometric reasoning"],"topics":["Agents","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.14397","pdf":"https://arxiv.org/pdf/2606.14397","project":null,"code":"https://github.com/gauntlet-benchmark/evaluation-harness","data":null,"hfPaper":"https://huggingface.co/papers/2606.14397"},"evidence":{"snippet":"To this end, we introduce GauntletBench, a web-based benchmark for evaluating agent generalisation in challenging scenarios, focusing on three underexplored capabilities (temporal perception, graphical understanding, and 3D reasoning), across five less-covered professional applications (Video Editor, Workflow Builder, 3D Modeller, Flight Analyser, and Circuit Designer), each with 20 vision-intensive tasks (100 in total).","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":19,"hfDailySubmittedAt":"2026-06-26T00:00:00.000Z","githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.14397"},"ranking":{"90d":{"score":28,"rank":252,"coverage":0.7,"confidence":"Medium"}},"description":"GauntletBench evaluates agent generalisation across five professional web applications with 100 vision-intensive tasks, probing temporal perception, graphical understanding, and 3D reasoning via automated objective scoring.","whyItMatters":"Existing agent benchmarks saturate and overlook harder capabilities; this benchmark reveals significant gaps in frontier agents, guiding development toward more robust real-world systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9993efbd5567a9dad8a4342b89529f7361b30868cc125412d1cfa395673fe155"},"motivation":"As agentic systems continue to evolve and are widely deployed in real-world scenarios, there is a growing demand to faithfully evaluate their capabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.14397","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_gaze-target-estimation-anywhere-with-conce_dea4f989","familyId":"bmf_f3392c2c6486","name":"Gaze-Co","oneLine":"Evaluates promptable gaze target estimation: given an image and a text or visual prompt identifying a subject, models output the subject's location, in/out-of-frame status, and a gaze target heatmap.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.11367","pdf":"https://arxiv.org/pdf/2608.11367","project":null,"code":"https://github.com/IrohXu/GazeAnywhere","data":"https://huggingface.co/datasets/IrohXu/Gaze-Co-Benchmark","hfPaper":null},"evidence":{"snippet":"We develop a scalable data engine to generate Gaze-Co (Gaze Estimation with Concepts), a dataset and benchmark of 120K high-quality, prompt-annotated image pairs.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":"2026-08-13T00:00:00.000Z","githubStars":22,"githubScope":"benchmark_repo","hfDatasetDownloads":327,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2608.11367"},"ranking":{"30d":{"score":44,"rank":41,"coverage":1.0,"confidence":"High","datasetDownloadRank":12,"datasetRankPopulation":30},"90d":{"score":43,"rank":132,"coverage":1.0,"confidence":"High","datasetDownloadRank":27,"datasetRankPopulation":66}},"description":"Evaluates promptable gaze target estimation: given an image and a text or visual prompt identifying a subject, models output the subject's location, in/out-of-frame status, and a gaze target heatmap.","whyItMatters":"Introduces concept-driven gaze estimation that removes brittle intermediate pipelines and provides reusable prompted splits across existing gaze datasets.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"93fb1e59ba20d21c1ac022a74a6f535cb68fe778281c318681bc05ba7eccdbcb"},"motivation":"Estimating human gaze targets from images in-the-wild is an important and formidable task.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with released dataset, evaluation code, and defined metrics (AUC, L2 distance, AP) on Hugging Face and GitHub.","canonicalNameSource":"abstract","canonicalNameEvidence":"Gaze-Co (Gaze Estimation with Concepts), a dataset and benchmark of 120K high-quality, prompt-annotated image pairs."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11367","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":72,"confidence":"Medium","horizon":"7d","reason":"CVPR 2026 paper, strong benchmark with open code and data, and novelty of promptable gaze estimation likely drive attention."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_fd6559b3d2319ec5","familyId":"catalog_family_fd6559b3d2319ec5","name":"GBA-Eval","oneLine":"An agentic coding benchmark that asks models to build a Game Boy Advance emulator from scratch and grades emulator behavior against procedural, audio, and gameplay tests.","description":"An agentic coding benchmark that asks models to build a Game Boy Advance emulator from scratch and grades emulator behavior against procedural, audio, and gameplay tests.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://gbaeval.com/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fd6559b3d2319ec5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gbaeval"}],"catalogSources":[{"catalog":"benchlm","sourceId":"gbaEval","url":"https://benchlm.ai/benchmarks/gbaeval","paperUrl":"https://gbaeval.com/","year":"2026","fullName":"GBA-Eval","format":"Overall emulator score","tasks":"27 emulator test cases","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_gbu-palm_d39e8a5c","familyId":"bmf_fcc9a8d5350b","name":"GBU-Palm","oneLine":"GBU-Palm is a large-scale multimodal video dataset for palm presentation attack detection, containing 21,326 videos from 105 subjects across six acquisition environments, including bona fide, Print, and Replay attacks, with 6,310 synchronized RGB-NIR samples. It evaluates video architectures under environment-matched and held-out-environment protocols, using metrics such as true accept, true reject, false accept, and false reject rates.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.14389","pdf":"https://arxiv.org/pdf/2608.14389","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.14389"},"evidence":{"snippet":"We present GBU-Palm, a large-scale multimodal video dataset and benchmark containing 21,326 videos from 105 subjects and 210 palms across six acquisition environments, including bona fide, Print, and Replay presentations, with 6,310 synchronized RGB-NIR samples.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14389"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GBU-Palm is a large-scale multimodal video dataset for palm presentation attack detection, containing 21,326 videos from 105 subjects across six acquisition environments, including bona fide, Print, and Replay attacks, with 6,310 synchronized RGB-NIR samples. It evaluates video architectures under environment-matched and held-out-environment protocols, using metrics such as true accept, true reject, false accept, and false reject rates.","whyItMatters":"Existing palm PAD datasets are limited by static imagery and restricted conditions, hindering systematic evaluation. GBU-Palm provides a unified benchmark to assess robustness across environments and modalities, helping practitioners choose architectures that generalize under environmental shift.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"aed987b576ae0994b262bdf02a421565e78b2615fb0504396c60b522b16f5f96"},"motivation":"Existing palm presentation attack detection (PAD) datasets are often limited by static imagery, restricted acquisition conditions, or insufficient multimodal video data, hindering systematic evaluation across environments, modalities, and attack types.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14389","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GBU-Palm Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.14389","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_gca-bench_71ae6cdc","familyId":"bmf_4566b765b276","name":"GCA-Bench","oneLine":"GCA-Bench evaluates robotic grasping in scenarios requiring scene-level reasoning and semantic constraints, comparing large foundation models on complex action tasks.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14341","pdf":"https://arxiv.org/pdf/2607.14341","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.14341"},"evidence":{"snippet":"To address this gap, we propose GCA-Bench, a benchmark featuring challenging \\textit{grasping with complex action} scenarios that involve both scene-level reasoning and semantic constraints.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14341"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GCA-Bench evaluates robotic grasping in scenarios requiring scene-level reasoning and semantic constraints, comparing large foundation models on complex action tasks.","whyItMatters":"The benchmark addresses a gap in grasping evaluation by moving beyond isolated visual pose detection to multi-step reasoning, offering a more realistic assessment for real-world robotic applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"901aa754e4aeb2d27f847ad43bae84f6e166e9977fa44754d2904f1f8793015d"},"motivation":"Robust robotic grasping remains a fundamental challenge for complex real-world applications.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14341","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_a2ed7249d0f2c4d1","familyId":"catalog_family_a2ed7249d0f2c4d1","name":"GDP.pdf","oneLine":"GDP.pdf is a knowledge-work vision benchmark that evaluates models on economically valuable professional tasks presented as visual documents (PDFs), testing document-based reasoning, chart and table interpretation, and problem solving without tools.","description":"GDP.pdf is a knowledge-work vision benchmark that evaluates models on economically valuable professional tasks presented as visual documents (PDFs), testing document-based reasoning, chart and table interpretation, and problem solving without tools.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","General","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/gdp-pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a2ed7249d0f2c4d1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gdp-pdf"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"gdp-pdf","url":"https://llm-stats.com/benchmarks/gdp-pdf","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","general","vision"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_5dec9212b4d600b7","familyId":"catalog_family_5dec9212b4d600b7","name":"GDP.pdf (no tools)","oneLine":"Professional document understanding over 100 real-world PDFs from ten domains.","description":"Professional document understanding over 100 real-world PDFs from ten domains.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5dec9212b4d600b7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gdppdf"}],"catalogSources":[{"catalog":"benchlm","sourceId":"gdpPdf","url":"https://benchlm.ai/benchmarks/gdppdf","paperUrl":"https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world","year":"2026","fullName":"GDP.pdf mean criteria pass rate without tools","format":"Mean criteria pass rate","tasks":"100 professional document prompts","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0c2e78c798a06faa","familyId":"catalog_family_0c2e78c798a06faa","name":"GDP.pdf (tools)","oneLine":"Professional document understanding with a container, standard libraries, and image cropping.","description":"Professional document understanding with a container, standard libraries, and image cropping.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0c2e78c798a06faa"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gdppdfwithtools"}],"catalogSources":[{"catalog":"benchlm","sourceId":"gdpPdfWithTools","url":"https://benchlm.ai/benchmarks/gdppdfwithtools","paperUrl":"https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world","year":"2026","fullName":"GDP.pdf mean criteria pass rate with tools","format":"Mean criteria pass rate with tools","tasks":"100 professional document prompts","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_gdpevo_5e650b40","familyId":"bmf_7fbe6f3635be","name":"GDPevo","oneLine":"GDPevo evaluates agent self-evolution on real business tasks across 24 groups (240 tasks) in domains like CRM, ERP, finance, and healthcare. It uses rule hybridization to attribute test-time gains to training experience, with held-out test tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Self-Evolution","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03764","pdf":"https://arxiv.org/pdf/2608.03764","project":null,"code":"https://github.com/Prism-Shadow/GDPevo","data":null,"hfPaper":"https://huggingface.co/papers/2608.03764"},"evidence":{"snippet":"We present GDPevo, an evolution-native benchmark grounded in GDP-related enterprise workflows, together with the fully automated data pipeline that generates it.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":27,"hfDailySubmittedAt":"2026-08-06T00:00:00.000Z","githubStars":62,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03764"},"ranking":{"30d":{"score":59,"rank":11,"coverage":0.85,"confidence":"High"},"90d":{"score":55,"rank":36,"coverage":0.7,"confidence":"Medium"}},"description":"GDPevo evaluates agent self-evolution on real business tasks across 24 groups (240 tasks) in domains like CRM, ERP, finance, and healthcare. It uses rule hybridization to attribute test-time gains to training experience, with held-out test tasks.","whyItMatters":"Existing benchmarks lack attribution of gains to training experience and face data contamination. This benchmark provides an automated pipeline for evolving benchmark instances and measures self-evolution ability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2d42f74ab15e994c8e08c218936059ee891d31c0d96df64449f638381afcabea"},"motivation":"Agent self-evolution updates an agent's persistent state from prior experience and reuses it to solve related tasks more effectively.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03764","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Prism-Shadow","organizationType":"community","sourceUrl":"https://github.com/Prism-Shadow/GDPevo","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_8aacade59f3852c9","familyId":"catalog_family_8aacade59f3852c9","name":"GDPval","oneLine":"GDPval is an OpenAI benchmark evaluating AI models on economically valuable, real-world knowledge-work tasks spanning many professional occupations and industries.","description":"GDPval is an OpenAI benchmark evaluating AI models on economically valuable, real-world knowledge-work tasks spanning many professional occupations and industries.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Legal","Reasoning","Finance","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/gdpval","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8aacade59f3852c9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gdpval"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"gdpval","url":"https://llm-stats.com/benchmarks/gdpval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["legal","reasoning","finance","general","agents"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_3a70d85f10b95f69","familyId":"catalog_family_3a70d85f10b95f69","name":"GDPval rubrics","oneLine":"GDPval-Rubrics evaluates AI model performance on economically valuable knowledge work tasks drawn from the public GDPval dataset. It uses pointwise scoring based on public rubrics, with the environment aligned to the GDPval-AA scaffolding.","description":"GDPval-Rubrics evaluates AI model performance on economically valuable knowledge work tasks drawn from the public GDPval dataset. It uses pointwise scoring based on public rubrics, with the environment aligned to the GDPval-AA scaffolding.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Agentic","Legal","Reasoning","Finance","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/MiniMaxAI/MiniMax-M3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3a70d85f10b95f69"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gdpvalrubrics"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gdpval-rubrics"}],"catalogSources":[{"catalog":"benchlm","sourceId":"gdpvalRubrics","url":"https://benchlm.ai/benchmarks/gdpvalrubrics","paperUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","year":"2026","fullName":"GDPval rubrics","format":"Rubric score","tasks":"Economically valuable work tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"gdpval-rubrics","url":"https://llm-stats.com/benchmarks/gdpval-rubrics","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","legal","reasoning","finance","general","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_22b7ac366fb64f0a","familyId":"catalog_family_22b7ac366fb64f0a","name":"GDPval-AA","oneLine":"GDPval-AA evaluates AI agents on economically valuable professional knowledge-work tasks and reports performance as an Elo score.","description":"GDPval-AA evaluates AI agents on economically valuable professional knowledge-work tasks and reports performance as an Elo score.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Agentic","Multimodalgrounded","Legal","Reasoning","Finance","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_22b7ac366fb64f0a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gdpvalaa"},{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gdpvalaanormalized"},{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gdpvalaa"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gdpval-aa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"gdpvalAa","url":"https://benchlm.ai/benchmarks/gdpvalaa","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"GDPval-AA","format":"Elo","tasks":"Agentic real-world work tasks","successorKey":null},{"catalog":"benchlm","sourceId":"gdpvalAaNormalized","url":"https://benchlm.ai/benchmarks/gdpvalaanormalized","paperUrl":"https://artificialanalysis.ai/models/grok-4-3","year":"2026","fullName":"GDPval-AA normalized","format":"Normalized score","tasks":"Economically valuable tasks","successorKey":null},{"catalog":"benchlm","sourceId":"gdpvalAa","url":"https://benchlm.ai/benchmarks/gdpvalaa","paperUrl":"https://www.minimax.io/news/minimax-m27-en","year":"2026","fullName":"GDPval-AA","format":"ELO-style office benchmark","tasks":"Professional office delivery","successorKey":null},{"catalog":"llm-stats","sourceId":"gdpval-aa","url":"https://llm-stats.com/benchmarks/gdpval-aa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","multimodalGrounded","legal","reasoning","finance","general","agents"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_44e326eb9a9e1060","familyId":"catalog_family_44e326eb9a9e1060","name":"GDPval-MM","oneLine":"GDPval-MM is the multimodal variant of the GDPval benchmark, evaluating AI model performance on real-world economically valuable tasks that require processing and generating multimodal content including documents, slides, diagrams, spreadsheets, images, and other professional deliverables across diverse industries.","description":"GDPval-MM is the multimodal variant of the GDPval benchmark, evaluating AI model performance on real-world economically valuable tasks that require processing and generating multimodal content including documents, slides, diagrams, spreadsheets, images, and other professional deliverables across diverse industries.","area":"Multimodal","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Finance","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/gdpval-mm","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_44e326eb9a9e1060"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gdpval-mm"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"gdpval-mm","url":"https://llm-stats.com/benchmarks/gdpval-mm","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","finance","general"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_geb-bench_e0eb6ac0","familyId":"bmf_d4f2e362a4ae","name":"GEB-Bench","oneLine":"GEB-Bench evaluates models on abstract structural motifs (e.g., self-reference, strange loops) presented in multiple modalities (natural scenes, folk stories, math theorems, programmatic skeletons) with tasks probing cross-modal mapping.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04111","pdf":"https://arxiv.org/pdf/2608.04111","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.04111"},"evidence":{"snippet":"We introduce GEB-Bench, a benchmark whose unit is an abstract structural motif--self-reference, a strange loop, a Mobius twist--in the spirit of Godel, Escher, Bach.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04111"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GEB-Bench evaluates models on abstract structural motifs (e.g., self-reference, strange loops) presented in multiple modalities (natural scenes, folk stories, math theorems, programmatic skeletons) with tasks probing cross-modal mapping.","whyItMatters":"The benchmark aims to measure abstraction and cross-modal transfer, which are foundational for general intelligence but not addressed by current benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"899f4d4c01ea08d02222f773e78e95652c08b561e83b505fb17cc433c106a6e1"},"motivation":"Can a model look at a river delta and a lightning bolt and see that they share a structure?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04111","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_geneb_31b4a4f4","familyId":"bmf_85ca5f7ab2ea","name":"GENEB","oneLine":"GENEB evaluates frozen representations from 40 genomic foundation models across 100 DNA classification tasks in 13 functional categories, using a unified linear probing protocol with full-data, 10-shot, and 1-shot regimes. Primary metric is Matthews correlation coefficient, with rankings at overall, category, and task levels.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04525","pdf":"https://arxiv.org/pdf/2606.04525","project":null,"code":"https://github.com/darlednik/GENEB","data":null,"hfPaper":"https://huggingface.co/papers/2606.04525"},"evidence":{"snippet":"We introduce GENEB, a large-scale diagnostic benchmark that evaluates frozen representations from 40 genomic foundation models across 100 tasks spanning 13 functional categories under a unified probing-based protocol, including few-shot regimes.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":49,"hfDailySubmittedAt":"2026-06-08T00:00:00.000Z","githubStars":45,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04525"},"ranking":{"90d":{"score":54,"rank":42,"coverage":0.7,"confidence":"Medium"}},"description":"GENEB evaluates frozen representations from 40 genomic foundation models across 100 DNA classification tasks in 13 functional categories, using a unified linear probing protocol with full-data, 10-shot, and 1-shot regimes. Primary metric is Matthews correlation coefficient, with rankings at overall, category, and task levels.","whyItMatters":"Genomic model comparisons are fragmented across incompatible protocols, making claims of superiority unreliable. GENEB provides a controlled, multi-task reference for category-aware model selection, revealing that aggregate leaderboards are unstable and scale gains are inconsistent.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0d0ed786690b2a64905443e202f1633195faa91195eaebe0bcbc6a44b65c2915"},"motivation":"Progress in genomic foundation models is difficult to assess due to fragmented benchmarks, incompatible evaluation protocols, and task-specific reporting.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026","evidence":"Accepted to ICML 2026","evidenceUrl":"https://arxiv.org/abs/2606.04525","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026","reviewStatus":"accepted","decisionRaw":"Accepted to ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.04525","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to ICML 2026","level":"author-claim"}]}],"publishers":[{"name":"GENEB team","organizationType":"academic-lab","sourceUrl":"https://github.com/darlednik/GENEB","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_4180a59b1ba29e4a","familyId":"catalog_family_4180a59b1ba29e4a","name":"GeneBench","oneLine":"GeneBench is an evaluation focused on multi-stage scientific data analysis in genetics and quantitative biology. Tasks require reasoning about ambiguous or noisy data with minimal supervisory guidance, addressing realistic obstacles such as hidden confounders or QC failures, and correctly implementing and interpreting modern statistical methods.","description":"GeneBench is an evaluation focused on multi-stage scientific data analysis in genetics and quantitative biology. Tasks require reasoning about ambiguous or noisy data with minimal supervisory guidance, addressing realistic obstacles such as hidden confounders or QC failures, and correctly implementing and interpreting modern statistical methods.","area":"Agents & Tool Use","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Reasoning","Science","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/genebench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4180a59b1ba29e4a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/genebench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"genebench","url":"https://llm-stats.com/benchmarks/genebench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","science","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_0a83fae6a9b340ef","familyId":"catalog_family_0a83fae6a9b340ef","name":"GeneBench-Pro","oneLine":"GeneBench-Pro is a research-level benchmark of 129 multi-stage computational-biology problems spanning genomics, quantitative biology, and translational biomedicine. Each problem gives the agent a messy dataset, brief context, and a target estimand, and requires navigating dependent inferential decision points to reach a verifiable answer.","description":"GeneBench-Pro is a research-level benchmark of 129 multi-stage computational-biology problems spanning genomics, quantitative biology, and translational biomedicine. Each problem gives the agent a messy dataset, brief context, and a target estimand, and requires navigating dependent inferential decision points to reach a verifiable answer.","area":"Agents & Tool Use","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Reasoning","Science","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://cdn.openai.com/pdf/21938268-21af-442f-af93-3b2249afb241/genebench-pro.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0a83fae6a9b340ef"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/genebenchpro"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/genebench-pro"}],"catalogSources":[{"catalog":"benchlm","sourceId":"geneBenchPro","url":"https://benchlm.ai/benchmarks/genebenchpro","paperUrl":"https://cdn.openai.com/pdf/21938268-21af-442f-af93-3b2249afb241/genebench-pro.pdf","year":2026,"fullName":"GeneBench-Pro","format":"Eval-level pass rate across dependent analysis decisions","tasks":"129 genomics statistical-analysis workflows","successorKey":null},{"catalog":"llm-stats","sourceId":"genebench-pro","url":"https://llm-stats.com/benchmarks/genebench-pro","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","science","agents"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_generative-embedding-benchmark_a087748d","familyId":"bmf_e7d9564105e1","name":"Generative Embedding Benchmark","oneLine":"GEB evaluates embeddings by measuring answer-relevant content recoverable by a decoder, using a visual question-answering dataset with development and test splits, and scoring answer quality.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.06972","pdf":"https://arxiv.org/pdf/2608.06972","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.06972"},"evidence":{"snippet":"To address this gap, we introduce the Generative Embedding Benchmark (GEB), in which a decoder answers questions using only a frozen embedding and question text, without access to the original image or intermediate visual features.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06972"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"GEB evaluates embeddings by measuring answer-relevant content recoverable by a decoder, using a visual question-answering dataset with development and test splits, and scoring answer quality.","whyItMatters":"Fills a gap by measuring generative information in embeddings, which is not captured by separability-based benchmarks, providing insight into information preservation for downstream generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d69187a6af41b75f5dc5ec62382306abe11b189f588a69c2178e6ba24b19450f"},"motivation":"Embeddings have emerged as a standard representational interface linking foundation models with downstream systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06972","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_geniac-secbench_393dfc82","familyId":"bmf_3e1ff571a6a3","name":"GenIaC-SecBench","oneLine":"Large language models are increasingly used to author Infrastructure-as-Code (IaC), where a single insecure default can be deployed directly into production.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.28021v1","pdf":"https://arxiv.org/pdf/2608.28021v1","project":null,"code":"https://github.com/AnimeshShaw/GenIaC-SecBench","data":"https://huggingface.co/datasets/AnimeshShaw/GenIaC-SecBench","hfPaper":null},"evidence":{"snippet":"We introduce GenIaC-SecBench, a benchmark of 100 deployment scenarios stratified by architectural complexity, evaluated across 12 model configurations from four vendors, producing 1,196 IaC artifacts scanned by three independent policy engines (Checkov, Trivy, KICS).","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":72,"hfDatasetLikes":1},"source":{"type":"arxiv","id":"2608.28021"},"ranking":{"30d":{"score":7,"rank":169,"coverage":1.0,"confidence":"High","datasetDownloadRank":23,"datasetRankPopulation":30},"90d":{"score":11,"rank":409,"coverage":1.0,"confidence":"High","datasetDownloadRank":51,"datasetRankPopulation":66}},"motivation":"Large language models are increasingly used to author Infrastructure-as-Code (IaC), where a single insecure default can be deployed directly into production.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T07:23:28.296315Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.28021v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_geo-bench_c7e53a2d","familyId":"bmf_cbe17d8c1c26","name":"GEO-Bench","oneLine":"GEO-Bench evaluates ranking manipulation attacks in generative engine optimization under a unified protocol, measuring effectiveness and stealth across five datasets against a fixed ranker, with open-source attack implementations.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.29107","pdf":"https://arxiv.org/pdf/2605.29107","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29107"},"evidence":{"snippet":"We present GEO-Bench, a benchmark that evaluates GEO ranking-manipulation attacks under one protocol.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29107"},"ranking":{},"description":"GEO-Bench evaluates ranking manipulation attacks in generative engine optimization under a unified protocol, measuring effectiveness and stealth across five datasets against a fixed ranker, with open-source attack implementations.","whyItMatters":"Addresses the lack of standardized comparison in GEO attack research, enabling direct comparison across attack paradigms and supporting detection development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fbc9be61c8075dbaf02cc15489fa211cf86bd400e745dfc5be69d6607aff53de"},"motivation":"Large language models (LLMs) increasingly rank products, documents, and recommendations for user queries, which makes manipulating these rankings a growing concern for fairness and information integrity.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29107","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_geodisaster_9f255b61","familyId":"bmf_4fb86bec6f3a","name":"GeoDisaster","oneLine":"GeoDisaster is an operational geospatial disaster reasoning benchmark with 2,921 instances across 43 question types and five task families, integrating EO/GIS evidence and grounding answers in executable geospatial workflows.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.17246","pdf":"https://arxiv.org/pdf/2606.17246","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.17246"},"evidence":{"snippet":"We introduce GeoDisaster, an operational geospatial disaster reasoning benchmark with 2,921 verified instances across 43 question types and five task families: deforestation monitoring, multi-hazard analysis, building-damage assessment, flood-safe routing, and Sentinel-1 SAR flood monitoring.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.17246"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GeoDisaster is an operational geospatial disaster reasoning benchmark with 2,921 instances across 43 question types and five task families, integrating EO/GIS evidence and grounding answers in executable geospatial workflows.","whyItMatters":"GeoDisaster addresses the gap in evaluating tool-grounded spatial reasoning and structured decision-making for disaster response, offering a potential standard for assessing operational geo-intelligence in RS-VLMs and agentic systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d1405a2b503292a1256f72bd3557256f7ac39b107c88710272053284cc729e38"},"motivation":"Remote-sensing vision-language models (RS-VLMs) have advanced Earth-observation analysis toward visual interpretation and instruction-following, yet fall short of operational geo-intelligence, which demands tool-grounded spatial reasoning and structured, evidence-backed decisions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.17246","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_geofidelity-bench_953b4352","familyId":"bmf_99a8cc850143","name":"GeoFidelity-Bench","oneLine":"GeoFidelity-Bench evaluates segment-level geographic fidelity in text-to-image street-view generation using 7,117 Mapillary images across 109 road segments in 25 cities.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.23669","pdf":"https://arxiv.org/pdf/2606.23669","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.23669"},"evidence":{"snippet":"We introduce GeoFidelity-Bench, a reference-panel benchmark for segment-conditioned geographic fidelity in street-view generation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.23669"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GeoFidelity-Bench evaluates segment-level geographic fidelity in text-to-image street-view generation using 7,117 Mapillary images across 109 road segments in 25 cities.","whyItMatters":"It addresses the gap between city-plausible and segment-accurate street-view generation, offering a protocol for comparing models on local geographic discrimination.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"49d5d910f8e12f35eec199a7cdbdc82ffa932b76c83c7e102f095d8bbe9eaea2"},"motivation":"Text-to-image models can generate visually plausible city streets, but whether their outputs correspond to a requested road segment rather than a generic city prior remains unclear.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.23669","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_geoid-flood_b555af53","familyId":"bmf_983fa02568b2","name":"GEOID-Flood","oneLine":"GEOID-Flood is a large-scale multi-modal benchmark for flood segmentation with over 14,000 tiles, co-registered SAR and optical data, and validated labels.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02315","pdf":"https://arxiv.org/pdf/2608.02315","project":null,"code":"https://github.com/links-ads/geoid-flood","data":null,"hfPaper":"https://huggingface.co/papers/2608.02315"},"evidence":{"snippet":"We introduce GEOID-Flood, a large-scale multi-modal flood segmentation benchmark, derived from Copernicus Emergency Management Service activations, spanning 219 events across 65 countries over ten years.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":"2026-08-04T00:00:00.000Z","githubStars":12,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02315"},"ranking":{"30d":{"score":34,"rank":62,"coverage":0.85,"confidence":"High"},"90d":{"score":37,"rank":170,"coverage":0.7,"confidence":"Medium"}},"description":"GEOID-Flood is a large-scale multi-modal benchmark for flood segmentation with over 14,000 tiles, co-registered SAR and optical data, and validated labels.","whyItMatters":"Supports evaluation of geospatial foundation models on flood mapping, enabling transfer studies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f40b62e3c40d49e9dda267f58975c5d3aac5c567ea961885abece2e007bd3a70"},"motivation":"Geospatial foundation models aim to learn representations that transfer across regions and sensors, yet evaluating them on specific tasks requires large, high-quality, multi-modal benchmarks that measure how well such models extract value from data.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026 - Terrabytes II Workshop, 23 pages","evidence":"Accepted at ECCV 2026 - Terrabytes II Workshop, 23 pages","evidenceUrl":"https://arxiv.org/abs/2608.02315","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026 - Terrabytes II Workshop, 23 pages","reviewStatus":"accepted","decisionRaw":"Accepted at ECCV 2026 - Terrabytes II Workshop, 23 pages","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.02315","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at ECCV 2026 - Terrabytes II Workshop, 23 pages","level":"author-claim"}]}],"publishers":[{"name":"links-ads","organizationType":"academic-lab","sourceUrl":"https://github.com/links-ads/geoid-flood","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_geonatureagent-benchmark_854d9ae1","familyId":"bmf_245d2d78860d","name":"GeoNatureAgent Benchmark","oneLine":"GeoNatureAgent Benchmark evaluates LLM agents on environmental geospatial analysis through structured tool calls to a self-hostable API. It includes 93 tasks across 18 categories, with metrics for accuracy and cost.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.12821","pdf":"https://arxiv.org/pdf/2606.12821","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.12821"},"evidence":{"snippet":"We introduce the GeoNatureAgent Benchmark, the first benchmark for environmental analysis agents that operate via structured tool calls to a production-style geospatial API.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.12821"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GeoNatureAgent Benchmark evaluates LLM agents on environmental geospatial analysis through structured tool calls to a self-hostable API. It includes 93 tasks across 18 categories, with metrics for accuracy and cost.","whyItMatters":"There is a lack of benchmarks for agentic geospatial workflows, and this benchmark provides a realistic API-based environment, enabling assessment of tool-use reasoning and cost-efficiency trade-offs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"258c8d828d24cdc11b7d02349ed26a10fff958e2f5deea883d5a0db021929c5a"},"motivation":"Environmental scientists spend disproportionate effort on data wrangling rather than analysis, and AI agents that automate geospatial workflows remain unvalidated: no benchmark evaluates agents operating through structured tool calling against real APIs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.12821","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_geot2v-bench_948c1065","familyId":"bmf_c3ac22233fcc","name":"GeoT2V-Bench","oneLine":"GeoT2V-Bench evaluates 3D consistency in camera-prompted text-to-video models via 3D reconstruction, using metrics like static rendering error and flow agreement.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.24829","pdf":"https://arxiv.org/pdf/2606.24829","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.24829"},"evidence":{"snippet":"We introduce GeoT2V-Bench, a reconstruction-based diagnostic benchmark for evaluating whether camera-prompted T2V clips can support explicit rigid 3D reconstruction.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24829"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GeoT2V-Bench evaluates 3D consistency in camera-prompted text-to-video models via 3D reconstruction, using metrics like static rendering error and flow agreement.","whyItMatters":"Assesses whether generated videos can support explicit rigid 3D reconstruction, providing a diagnostic tool for a critical limitation of T2V models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"15f09e26d6e0ac8fd9d968007cc0ce0064cd932296321c4884429740de727f04"},"motivation":"Camera-prompted text-to-video (T2V) models are increasingly used to synthesize virtual camera captures, such as orbiting objects or moving through static scenes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24829","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_3892a6c676ac67e7","familyId":"catalog_family_3892a6c676ac67e7","name":"Gert Labs","oneLine":"A game-environment benchmark that evaluates AI models in novel games covering strategic planning, resource management, spatial reasoning, cooperation, and theory of mind.","description":"A game-environment benchmark that evaluates AI models in novel games covering strategic planning, resource management, spatial reasoning, cooperation, and theory of mind.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://gertlabs.com/rankings","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3892a6c676ac67e7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gertlabs"}],"catalogSources":[{"catalog":"benchlm","sourceId":"gertLabs","url":"https://benchlm.ai/benchmarks/gertlabs","paperUrl":"https://gertlabs.com/rankings","year":"2026","fullName":"Gert Labs Composite Game Benchmark","format":"Composite game leaderboard","tasks":"Novel game environments","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_38e7b95132e4409c","familyId":"catalog_family_38e7b95132e4409c","name":"GiantSteps Tempo","oneLine":"A dataset for tempo estimation in electronic dance music containing 664 2-minute audio previews from Beatport, annotated from user corrections for evaluating automatic tempo estimation algorithms.","description":"A dataset for tempo estimation in electronic dance music containing 664 2-minute audio previews from Beatport, annotated from user corrections for evaluating automatic tempo estimation algorithms.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/giantsteps-tempo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_38e7b95132e4409c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/giantsteps-tempo"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"giantsteps-tempo","url":"https://llm-stats.com/benchmarks/giantsteps-tempo","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["audio"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_gisagentbench_880052a5","familyId":"bmf_6d527eda7e33","name":"GISAgentBench","oneLine":"GISAgentBench evaluates LLM agents on multi-step GIS tasks from practitioner sources, with 349 tasks and executable reference trajectories.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.01645","pdf":"https://arxiv.org/pdf/2608.01645","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01645"},"evidence":{"snippet":"To address this gap, we introduce GISAgentBench, a benchmark of 349 multi-step GIS tasks curated from GIS Stack Exchange and instantiated on real public data across six selected geographic areas of interest.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01645"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GISAgentBench evaluates LLM agents on multi-step GIS tasks from practitioner sources, with 349 tasks and executable reference trajectories.","whyItMatters":"Provides deterministic evaluation with ground truth outputs, addressing limitations of surrogate signals.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"073a78b19cc0e94ac668cf91be2d364711fdd4c49dbf2626383828d202f64bdf"},"motivation":"Geographic Information System (GIS) professionals rely on multi-step spatial analysis workflows to support decision-making in urban planning, disaster response, and environmental monitoring.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.01645","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_gischolarbench_a6158b85","familyId":"bmf_373e6eeda3f2","name":"GIScholarBench","oneLine":"GIScholarBench evaluates LLM overconfidence in GIS research across three tasks: metadata retrieval, literature linking, and research direction generation, using 10,865 papers from 25 GIScience journals.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.IR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08036","pdf":"https://arxiv.org/pdf/2606.08036","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.08036"},"evidence":{"snippet":"To examine this issue, we introduce GIScholarBench, a benchmark built from 10,865 papers published in 25 core GIScience journals between 2020 and 2025.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08036"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GIScholarBench evaluates LLM overconfidence in GIS research across three tasks: metadata retrieval, literature linking, and research direction generation, using 10,865 papers from 25 GIScience journals.","whyItMatters":"It addresses the need for benchmarks that assess factual accuracy and overconfidence in scholarly AI applications, providing a basis for evaluating LLM reliability in research workflows.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1db69c2d5bb190d6d8d9398e0a81157fa3e6add49640ed2634fca67489141dcc"},"motivation":"Large language models (LLMs) are increasingly used in academic research workflows, but scholarly tasks require high factual precision and therefore expose a key weakness: overconfidence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08036","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_1532170c64d8f2b9","familyId":"catalog_family_1532170c64d8f2b9","name":"Global PIQA","oneLine":"Global PIQA is a multilingual commonsense reasoning benchmark that evaluates physical interaction knowledge across 100 languages and cultures. It tests AI systems' understanding of physical world knowledge in diverse cultural contexts through multiple choice questions about everyday situations requiring physical commonsense.","description":"Global PIQA is a multilingual commonsense reasoning benchmark that evaluates physical interaction knowledge across 100 languages and cultures. It tests AI systems' understanding of physical world knowledge in diverse cultural contexts through multiple choice questions about everyday situations requiring physical commonsense.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Physics","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/global-piqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1532170c64d8f2b9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/global-piqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"global-piqa","url":"https://llm-stats.com/benchmarks/global-piqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["physics","reasoning","general"],"catalogModelCount":13,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_69b9582b979cfa0a","familyId":"catalog_family_69b9582b979cfa0a","name":"Global-MMLU","oneLine":"A comprehensive multilingual benchmark covering 42 languages that addresses cultural and linguistic biases in evaluation, with improved translation quality and culturally sensitive question subsets.","description":"A comprehensive multilingual benchmark covering 42 languages that addresses cultural and linguistic biases in evaluation, with improved translation quality and culturally sensitive question subsets.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/global-mmlu","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_69b9582b979cfa0a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/global-mmlu"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"global-mmlu","url":"https://llm-stats.com/benchmarks/global-mmlu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning","general"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_885d8f9fd59e8815","familyId":"catalog_family_885d8f9fd59e8815","name":"Global-MMLU-Lite","oneLine":"A lightweight version of Global MMLU benchmark that evaluates language models across multiple languages while addressing cultural and linguistic biases in multilingual evaluation.","description":"A lightweight version of Global MMLU benchmark that evaluates language models across multiple languages while addressing cultural and linguistic biases in multilingual evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/global-mmlu-lite","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_885d8f9fd59e8815"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/global-mmlu-lite"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"global-mmlu-lite","url":"https://llm-stats.com/benchmarks/global-mmlu-lite","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning","general"],"catalogModelCount":15,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_globaldentbench_f42d9d35","familyId":"bmf_6790f75fc2f9","name":"GlobalDentBench","oneLine":"GlobalDentBench evaluates LLM clinical reasoning in dentistry with 8,978 expert-validated questions across 14 specialties and 88 countries, covering multiple-choice, short-answer, and case-based formats at three reasoning levels.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.24636","pdf":"https://arxiv.org/pdf/2605.24636","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24636"},"evidence":{"snippet":"Here we introduce GlobalDentBench, the first multinational dental benchmark, featuring a taxonomy that encompasses 14 dental specialties across 88 countries and regions spanning six continents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24636"},"ranking":{},"description":"GlobalDentBench evaluates LLM clinical reasoning in dentistry with 8,978 expert-validated questions across 14 specialties and 88 countries, covering multiple-choice, short-answer, and case-based formats at three reasoning levels.","whyItMatters":"It provides a multinational dental benchmark with expert calibration to assess knowledge recall, routine and individualized reasoning, revealing safety risks in LLM clinical recommendations and supporting rigorous validation before deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b3bfc579c8c91cd9b50169e0a1580df569385a8933a4ee21a87069f1b071b5a2"},"motivation":"While large language models (LLMs) hold transformative potential for medicine, their reasoning robustness and safety in real-world clinical scenarios remain critically underexplored, particularly in dentistry.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24636","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GlobalDentBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2605.24636","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_globeaudio_abb3482b","familyId":"bmf_575d54451a6e","name":"GlobeAudio","oneLine":"GlobeAudio evaluates audio-language models on naturalistic audio understanding across six languages. It includes 5,637 multiple-choice questions with naturally occurring audio, testing auditory reasoning and cultural interpretation.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08194","pdf":"https://arxiv.org/pdf/2606.08194","project":null,"code":null,"data":"https://huggingface.co/datasets/iNLP-Lab/GlobeAudio","hfPaper":"https://huggingface.co/papers/2606.08194"},"evidence":{"snippet":"To bridge this gap, we propose GlobeAudio, a multilingual and multicultural benchmark designed to evaluate naturalistic audio understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":4,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":48,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2606.08194"},"ranking":{"90d":{"score":41,"rank":140,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":63,"datasetRankPopulation":66}},"description":"GlobeAudio evaluates audio-language models on naturalistic audio understanding across six languages. It includes 5,637 multiple-choice questions with naturally occurring audio, testing auditory reasoning and cultural interpretation.","whyItMatters":"Existing benchmarks lack linguistic and cultural authenticity and acoustic realism. GlobeAudio addresses this gap for comparison of models under real-world conditions, highlighting performance differences, especially for open-source models and low-resource languages.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"890c02f52ae757e4b09c10cf947ceb93b2f83c1d8dc44b5400724c346c227a3f"},"motivation":"Large Audio-Language Models (LALMs) integrate audio perception and language understanding within a unified framework, enabling a wide range of real-world applications.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08194","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"iNLP-Lab","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/datasets/iNLP-Lab/GlobeAudio","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_glucofm-bench_d664b863","familyId":"bmf_b3ee767d25c4","name":"GlucoFM-Bench","oneLine":"GlucoFM-Bench evaluates time-series foundation models for blood glucose forecasting across 15 datasets, protocols including zero-shot, few-shot, and full-shot, with metrics like RMSE.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06881","pdf":"https://arxiv.org/pdf/2606.06881","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.06881"},"evidence":{"snippet":"To bridge this gap, we present GlucoFM-Bench, a comprehensive benchmark evaluating state-of-the-art TSFMs alongside supervised deep learning models for blood glucose forecasting.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06881"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GlucoFM-Bench evaluates time-series foundation models for blood glucose forecasting across 15 datasets, protocols including zero-shot, few-shot, and full-shot, with metrics like RMSE.","whyItMatters":"Provides standardized evaluation for glucose forecasting, clarifying when TSFMs outperform supervised models, aiding model selection in diabetes management.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"833272dc81a4a5b160f59f334f1f504de215140c6a26a52347b51c50768a85e3"},"motivation":"Blood glucose forecasting models are foundational for modern diabetes management systems, as reliable short-term predictions can enable proactive interventions, support automated insulin delivery, and reduce the risk of hypo- and hyperglycemic events.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.06881","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0f22e6c87bdca405","familyId":"catalog_family_0f22e6c87bdca405","name":"GMMLU","oneLine":"MMLU-style knowledge evaluation across 42 high- and low-resource languages.","description":"MMLU-style knowledge evaluation across 42 high- and low-resource languages.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multilingual"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2412.03304","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0f22e6c87bdca405"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gmmlu"}],"catalogSources":[{"catalog":"benchlm","sourceId":"gmmlu","url":"https://benchlm.ai/benchmarks/gmmlu","paperUrl":"https://arxiv.org/abs/2412.03304","year":"2024","fullName":"Global MMLU","format":"Average accuracy","tasks":"Knowledge questions across 42 languages","successorKey":null}],"catalogCategories":["multilingual"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_66d5dc0b03474e67","familyId":"catalog_family_66d5dc0b03474e67","name":"Gorilla Benchmark API Bench","oneLine":"APIBench, a comprehensive dataset of over 11,000 instruction-API pairs from HuggingFace, TorchHub, and TensorHub APIs for evaluating language models' ability to generate accurate API calls.","description":"APIBench, a comprehensive dataset of over 11,000 instruction-API pairs from HuggingFace, TorchHub, and TensorHub APIs for evaluating language models' ability to generate accurate API calls.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Code","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/gorilla-benchmark-api-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_66d5dc0b03474e67"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gorilla-benchmark-api-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"gorilla-benchmark-api-bench","url":"https://llm-stats.com/benchmarks/gorilla-benchmark-api-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","code","tool calling"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_6513c0b4e0e4c6ed","familyId":"catalog_family_6513c0b4e0e4c6ed","name":"GovReport","oneLine":"A long document summarization dataset consisting of reports from government research agencies including Congressional Research Service and U.S. Government Accountability Office, with significantly longer documents and summaries than other datasets.","description":"A long document summarization dataset consisting of reports from government research agencies including Congressional Research Service and U.S. Government Accountability Office, with significantly longer documents and summaries than other datasets.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Summarization"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/govreport","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6513c0b4e0e4c6ed"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/govreport"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"govreport","url":"https://llm-stats.com/benchmarks/govreport","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","summarization"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"lib_gpqa","familyId":"family_gpqa","name":"GPQA","oneLine":"Established benchmark family · Knowledge & Reasoning.","area":"Knowledge & Reasoning","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Knowledge & Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2023-01-01","releaseDatePrecision":"year","firstRelease":{"year":2023,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2311.12022","pdf":null,"project":"https://github.com/idavidrein/gpqa","code":"https://github.com/idavidrein/gpqa","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_gpqa"},"ranking":{},"recordType":"family","aliases":["Graduate-Level Google-Proof Q&A"],"sourceAttribution":[{"role":"benchmark-paper","url":"https://arxiv.org/abs/2311.12022"}],"adoptionRefs":["openai-gpt5","anthropic-claude4","google-gemini25","deepseek-v3"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"},{"sourceId":"deepseek-v3","url":"https://github.com/deepseek-ai/DeepSeek-V3","provider":"DeepSeek","model":"DeepSeek-V3"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific","catalogSources":[{"catalog":"benchlm","sourceId":"gpqa","url":"https://benchlm.ai/benchmarks/gpqa","paperUrl":"https://arxiv.org/abs/2311.12022","year":"2023","fullName":"Graduate-Level Google-Proof Q&A","format":"Multiple choice questions","tasks":"448 questions","successorKey":null},{"catalog":"llm-stats","sourceId":"gpqa","url":"https://llm-stats.com/benchmarks/gpqa","datasetSlug":"gpqa","versionCount":1,"subsetCount":2,"rowCount":null,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":["knowledge","physics","reasoning","general","biology","chemistry"],"catalogModelCount":241,"catalogStarCount":5},{"id":"catalog_9b2670faedafe987","familyId":"catalog_family_9b2670faedafe987","name":"GPQA Biology","oneLine":"Biology subset of GPQA, containing challenging multiple-choice questions written by domain experts in biology. These Google-proof questions require graduate-level knowledge and reasoning.","description":"Biology subset of GPQA, containing challenging multiple-choice questions written by domain experts in biology. These Google-proof questions require graduate-level knowledge and reasoning.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Reasoning","General","Healthcare","Biology"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/gpqa-biology","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9b2670faedafe987"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gpqa-biology"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"gpqa-biology","url":"https://llm-stats.com/benchmarks/gpqa-biology","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","healthcare","biology"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_de8024fbe9575be8","familyId":"catalog_family_de8024fbe9575be8","name":"GPQA Chemistry","oneLine":"Chemistry subset of GPQA, containing challenging multiple-choice questions written by domain experts in chemistry. These Google-proof questions require graduate-level knowledge and reasoning.","description":"Chemistry subset of GPQA, containing challenging multiple-choice questions written by domain experts in chemistry. These Google-proof questions require graduate-level knowledge and reasoning.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Reasoning","Chemistry"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/gpqa-chemistry","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_de8024fbe9575be8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gpqa-chemistry"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"gpqa-chemistry","url":"https://llm-stats.com/benchmarks/gpqa-chemistry","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","chemistry"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"lib_gpqa_diamond","familyId":"family_gpqa","name":"GPQA Diamond","oneLine":"Established benchmark variant · Knowledge & Reasoning.","area":"Knowledge & Reasoning","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Knowledge & Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2023-01-01","releaseDatePrecision":"year","firstRelease":{"year":2023,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2311.12022","pdf":null,"project":"https://github.com/idavidrein/gpqa","code":"https://github.com/idavidrein/gpqa","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_gpqa_diamond"},"ranking":{},"recordType":"variant","aliases":["GPQA-Diamond"],"sourceAttribution":[{"role":"benchmark-definition","url":"https://github.com/idavidrein/gpqa"}],"adoptionRefs":["openai-gpt5","anthropic-claude4","google-gemini25"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"variantOf":"lib_gpqa","capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific","catalogSources":[{"catalog":"benchlm","sourceId":"gpqaDiamond","url":"https://benchlm.ai/benchmarks/gpqa-diamond","paperUrl":"https://arxiv.org/abs/2311.12022","year":"2023","fullName":"GPQA Diamond","format":"Multiple choice questions","tasks":"Expert-level science questions","successorKey":null}],"catalogCategories":["reasoning"],"catalogModelCount":0,"catalogStarCount":0},{"id":"catalog_8fcb539c7245dd97","familyId":"catalog_family_8fcb539c7245dd97","name":"GPQA Physics","oneLine":"Physics subset of GPQA, containing challenging multiple-choice questions written by domain experts in physics. These Google-proof questions require graduate-level knowledge and reasoning.","description":"Physics subset of GPQA, containing challenging multiple-choice questions written by domain experts in physics. These Google-proof questions require graduate-level knowledge and reasoning.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Physics","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/gpqa-physics","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8fcb539c7245dd97"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gpqa-physics"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"gpqa-physics","url":"https://llm-stats.com/benchmarks/gpqa-physics","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["physics","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_b69ff8537b15df25","familyId":"catalog_family_b69ff8537b15df25","name":"GPQA-D","oneLine":"A display-only GPQA Diamond reference from provider comparison charts.","description":"A display-only GPQA Diamond reference from provider comparison charts.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.arcee.ai/blog/trinity-large-thinking","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b69ff8537b15df25"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gpqa-diamond"}],"catalogSources":[{"catalog":"benchlm","sourceId":"gpqaDiamond","url":"https://benchlm.ai/benchmarks/gpqa-diamond","paperUrl":"https://www.arcee.ai/blog/trinity-large-thinking","year":"2026","fullName":"GPQA Diamond","format":"Multiple choice questions","tasks":"Graduate-level science questions","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_gptnt_3088efa1","familyId":"bmf_fa478faf966d","name":"GPTNT","oneLine":"GPTNT evaluates multimodal agents on real-time collaborative bomb defusal in the game Keep Talking and Nobody Explodes, requiring asynchronous communication under time pressure and information asymmetry. Success is measured by defusing procedurally generated bombs.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-06-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.28514","pdf":"https://arxiv.org/pdf/2606.28514","project":"https://gptnt.github.io","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.28514"},"evidence":{"snippet":"We introduce GPTNT, a benchmark built on the cooperative video game Keep Talking and Nobody Explodes, in which two agents must coordinate to defuse procedurally generated bomb puzzles against a live countdown.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.28514"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"GPTNT evaluates multimodal agents on real-time collaborative bomb defusal in the game Keep Talking and Nobody Explodes, requiring asynchronous communication under time pressure and information asymmetry. Success is measured by defusing procedurally generated bombs.","whyItMatters":"Current benchmarks isolate collaboration components; GPTNT captures time pressure, information asymmetry, and imperfect communication together, providing a realistic test for multimodal systems that current evaluations leave unmeasured.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9b8051d26fd905dc78c04a1d4c48f0f4e90b709ad10ee25058c269b4eb05572b"},"motivation":"Multimodal models are increasingly deployed to solve tasks collaboratively with humans or other artificial agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.28514","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GPTNT Project","organizationType":"academic-lab","sourceUrl":"https://gptnt.github.io","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_gpu-accelerated-ml-inference-benchmark_af49dfaf","familyId":"bmf_e63546d29f56","name":"GPU-Accelerated ML Inference Benchmark","oneLine":"Compares inference latency, throughput, and GPU utilization for PyTorch CPU, PyTorch CUDA, and TensorRT across batch sizes.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-23","firstSeenAt":"2026-08-24","recognitionConfidence":0.4,"links":{"report":"https://github.com/jyotydivya/GPU-Accelerated-ML-Inference-Benchmark","pdf":null,"project":null,"code":"https://github.com/jyotydivya/GPU-Accelerated-ML-Inference-Benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"GPU-Accelerated-ML-Inference-Benchmark # GPU-Accelerated ML Inference Benchmark This project demonstrates a benchmark comparing inference performance across different execution engines: 1.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:jyotydivya/gpu-accelerated-ml-inference-benchmark"},"ranking":{"30d":{"score":23,"rank":115,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":319,"coverage":0.55,"confidence":"Low"}},"description":"Compares inference latency, throughput, and GPU utilization for PyTorch CPU, PyTorch CUDA, and TensorRT across batch sizes.","whyItMatters":"Evaluates GPU-accelerated inference engines for practical deployment performance.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"c6900ba3d4ec1bca854c8418f6ec15e5daa859034e4b919a72379cfeb783dc42"},"motivation":"GPU-Accelerated-ML-Inference-Benchmark # GPU-Accelerated ML Inference Benchmark This project demonstrates a benchmark comparing inference performance across different execution engines: 1.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"No clearly defined stable scoring contract or independent public reuse path beyond a demonstration repository."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/jyotydivya/GPU-Accelerated-ML-Inference-Benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-24T07:04:32.475961Z"},"attentionForecast":{"score":3,"confidence":"Low","horizon":"7d","reason":"Single-repository demo with no independent paper or adoption signals, likely to attract minimal attention."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_grace_1b12b544","familyId":"bmf_e010fd1ce1ac","name":"GRACE","oneLine":"GRACE evaluates step-level faithfulness of chain-of-thought reasoning in context-grounded tasks, with human annotations and a taxonomy of error categories.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.16151","pdf":"https://arxiv.org/pdf/2606.16151","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.16151"},"evidence":{"snippet":"We introduce GRACE, the first human-annotated step-level faithfulness benchmark with a data-driven error taxonomy for context-grounded textual reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16151"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GRACE evaluates step-level faithfulness of chain-of-thought reasoning in context-grounded tasks, with human annotations and a taxonomy of error categories.","whyItMatters":"Step-level faithfulness assessment addresses the gap where response-level metrics miss localized reasoning failures, providing granular feedback for improving model reliability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"81a05d96ef78b0f815e196a14a5027cbad43d39f4c52185407e45feb16e05b4d"},"motivation":"Many reasoning tasks require models to reason over input context, from document-grounded question answering to rule-based deduction.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"EMNLP Main 2026","evidence":"Accepted at EMNLP Main 2026","evidenceUrl":"https://arxiv.org/abs/2606.16151","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"EMNLP Main 2026","reviewStatus":"accepted","decisionRaw":"Accepted at EMNLP Main 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.16151","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at EMNLP Main 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_grapharc_4326342c","familyId":"bmf_c87c449862ea","name":"GraphARC","oneLine":"GraphARC is a benchmark for abstract reasoning on graphs, generalizing ARC few-shot learning. No artifacts are provided in this article.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.31031","pdf":"https://arxiv.org/pdf/2605.31031","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.31031"},"evidence":{"snippet":"We introduce GraphARC, a benchmark for abstract reasoning on graph-structured data.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.31031"},"ranking":{},"description":"GraphARC is a benchmark for abstract reasoning on graphs, generalizing ARC few-shot learning. No artifacts are provided in this article.","whyItMatters":"It aims to evaluate relational reasoning in graph models, but lacks a public path for external evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4695562d7efdbd01afb65c168459e49005fcfe4e8bfbc5bd4f76e8f69a046413"},"motivation":"Relational reasoning lies at the heart of intelligence, but existing benchmarks are typically confined to formats such as grids or text.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"KDD 2026 Datasets and Benchmarks Track","evidence":"Accepted at KDD 2026 Datasets and Benchmarks Track","evidenceUrl":"https://arxiv.org/abs/2605.31031","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"KDD 2026 Datasets and Benchmarks Track","reviewStatus":"accepted","decisionRaw":"Accepted at KDD 2026 Datasets and Benchmarks Track","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2605.31031","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at KDD 2026 Datasets and Benchmarks Track","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_graphinfer-bench_ae7fff34","familyId":"bmf_be514e0e285e","name":"GraphInfer-Bench","oneLine":"GraphInfer-Bench evaluates LLMs on graph inference tasks where answers reside in no single node or path, covering five task types over six real-world graphs with 42,000 samples. Tasks include masked-node prediction, edge inference, theme description, outlier detection, and community partition.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11562","pdf":"https://arxiv.org/pdf/2606.11562","project":null,"code":"https://github.com/graphinfer/GraphInfer-Bench","data":"https://huggingface.co/datasets/graphinfer/graphinfer","hfPaper":"https://huggingface.co/papers/2606.11562"},"evidence":{"snippet":"We introduce GraphInfer-Bench, a benchmark for whether LLMs can perform this graph inference: producing an open-ended answer that no single node supports and no path retrieves.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":71,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2606.11562"},"ranking":{"90d":{"score":46,"rank":103,"coverage":0.3,"confidence":"Low","datasetDownloadRank":52,"datasetRankPopulation":66}},"description":"GraphInfer-Bench evaluates LLMs on graph inference tasks where answers reside in no single node or path, covering five task types over six real-world graphs with 42,000 samples. Tasks include masked-node prediction, edge inference, theme description, outlier detection, and community partition.","whyItMatters":"Targets an open capability gap in graph understanding: inference over joint neighborhood structure. Useful for diagnosing weaknesses in LLMs and GNNs for tasks like fraud detection, drug repurposing, and recommendation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"29792f151b116d3a20164d719648aac9e29b1499b1b1a65d41dbeef11782020c"},"motivation":"Graph analysis underlies many applications whose answers cannot be looked up in a single record or retrieved along a path: laundering rings, drug repurposing, user preference, and scientific theme are all inferred from a node together with its neighbourhood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11562","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GraphInfer-Bench team","organizationType":"community","sourceUrl":"https://github.com/graphinfer/GraphInfer-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_graphrarebench_9ff01b1d","familyId":"bmf_1c6412c93a36","name":"GraphRareBench","oneLine":"GraphRareBench evaluates phenotype-driven rare-disease ranking with 2,365 cases, 18,093 target-confounder pairs, and auditable evidence records.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["q-bio.QM"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.24878","pdf":"https://arxiv.org/pdf/2607.24878","project":null,"code":"https://github.com/GUI0609/GraphRareBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.24878"},"evidence":{"snippet":"We introduce GraphRareBench, a provenance-preserving benchmark containing 2,365 ontology-derived cases and 18,093 target-confounder pairs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24878"},"ranking":{"90d":{"score":23,"rank":365,"coverage":0.55,"confidence":"Low"}},"description":"GraphRareBench evaluates phenotype-driven rare-disease ranking with 2,365 cases, 18,093 target-confounder pairs, and auditable evidence records.","whyItMatters":"Offers a transparent, evidence-aware evaluation for diagnostic systems, measuring retrieval and hard-confounder discrimination in a provenance-preserving setting.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"32cc2d0e21e8e20b1d07804d0e1e065b90a29de8c970df54bfd51fbd0cd76f05"},"motivation":"Phenotype-driven diagnostic benchmarks usually report the rank of the reference disease, but they rarely reveal which plausible alternatives are ranked above it or what evidence a tool-using model examines before making its decision.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24878","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GraphRareBench team","organizationType":"academic-lab","sourceUrl":"https://github.com/GUI0609/GraphRareBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_graphverse_5c59831b","familyId":"bmf_52691295a1bc","name":"GraphVerse","oneLine":"GraphVerse evaluates multimodal large language models on visual graph reasoning, covering perception, reasoning, and text-based graph reasoning in single and paired image settings, with process-sensitive scoring.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.06769","pdf":"https://arxiv.org/pdf/2608.06769","project":null,"code":"https://github.com/sunyuanfu/GraphVerse","data":null,"hfPaper":"https://huggingface.co/papers/2608.06769"},"evidence":{"snippet":"To bridge the gap, we introduce GraphVerse, a unified benchmark that jointly evaluates perception, visual reasoning, and text-based graph reasoning in MLLMs under both single-image and paired-image settings.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06769"},"ranking":{"30d":{"score":23,"rank":156,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":360,"coverage":0.55,"confidence":"Low"}},"description":"GraphVerse evaluates multimodal large language models on visual graph reasoning, covering perception, reasoning, and text-based graph reasoning in single and paired image settings, with process-sensitive scoring.","whyItMatters":"Provides a unified benchmark for visual graph reasoning that goes beyond answer-only metrics, addressing gaps in existing evaluations and enabling assessment of reasoning quality.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"915f9e7086ce709ee6882ffcfd511aee15232a67b33142505e69d462bc62bd94"},"motivation":"Recent Multimodal Large Language Models (MLLMs) have achieved remarkable progress across diverse vision-language tasks, creating an urgent need for more challenging benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06769","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GraphVerse Team","organizationType":"academic-lab","sourceUrl":"https://github.com/sunyuanfu/GraphVerse","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_9b96b90782b4c763","familyId":"catalog_family_9b96b90782b4c763","name":"GraphWalks","oneLine":"GraphWalks is a synthetic multi-hop long-context reasoning benchmark in which a model is given an edge-list representation of a graph and must traverse it to find neighboring nodes (via breadth-first search) or parent nodes for a given start node. Performance is reported as F1 of the model-predicted answer set versus the ground truth.","description":"GraphWalks is a synthetic multi-hop long-context reasoning benchmark in which a model is given an edge-list representation of a graph and must traverse it to find neighboring nodes (via breadth-first search) or parent nodes for a given start node. Performance is reported as F1 of the model-predicted answer set versus the ground truth.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/graphwalks","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9b96b90782b4c763"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/graphwalks"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"graphwalks","url":"https://llm-stats.com/benchmarks/graphwalks","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_6d2eb34ec8f6b584","familyId":"catalog_family_6d2eb34ec8f6b584","name":"Graphwalks BFS 128K","oneLine":"A graph reasoning benchmark that evaluates language models' ability to perform breadth-first search (BFS) operations on graphs with context length over 128k tokens, testing long-context reasoning capabilities.","description":"A graph reasoning benchmark that evaluates language models' ability to perform breadth-first search (BFS) operations on graphs with context length over 128k tokens, testing long-context reasoning capabilities.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Spatial Reasoning","Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6d2eb34ec8f6b584"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/graphwalksbfs128k"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/graphwalks-bfs-<128k"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/graphwalks-bfs->128k"}],"catalogSources":[{"catalog":"benchlm","sourceId":"graphwalksBfs128k","url":"https://benchlm.ai/benchmarks/graphwalksbfs128k","paperUrl":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","year":"2026","fullName":"Graphwalks BFS 0K-128K","format":"Long-context graph reasoning","tasks":"Graph traversal tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"graphwalks-bfs-<128k","url":"https://llm-stats.com/benchmarks/graphwalks-bfs-<128k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false},{"catalog":"llm-stats","sourceId":"graphwalks-bfs->128k","url":"https://llm-stats.com/benchmarks/graphwalks-bfs->128k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","spatial reasoning","long context"],"catalogModelCount":11,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences","Long Context & Memory"],"domainScope":"general"},{"id":"catalog_53d58e9c28a37225","familyId":"catalog_family_53d58e9c28a37225","name":"Graphwalks BFS 1M","oneLine":"GraphWalks BFS variant evaluated on 1M-token contexts.","description":"GraphWalks BFS variant evaluated on 1M-token contexts.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","Spatial Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/graphwalks-bfs-1m","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_53d58e9c28a37225"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/graphwalks-bfs-1m"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"graphwalks-bfs-1m","url":"https://llm-stats.com/benchmarks/graphwalks-bfs-1m","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","spatial reasoning"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences","Long Context & Memory"],"domainScope":"general"},{"id":"catalog_2afafaa160f04bd4","familyId":"catalog_family_2afafaa160f04bd4","name":"Graphwalks Parents 128K","oneLine":"A graph reasoning benchmark that evaluates language models' ability to find parent nodes in graphs with context length over 128k tokens, testing long-context reasoning and graph structure understanding.","description":"A graph reasoning benchmark that evaluates language models' ability to find parent nodes in graphs with context length over 128k tokens, testing long-context reasoning and graph structure understanding.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Spatial Reasoning","Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2afafaa160f04bd4"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/graphwalksparents128k"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/graphwalks-parents-<128k"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/graphwalks-parents->128k"}],"catalogSources":[{"catalog":"benchlm","sourceId":"graphwalksParents128k","url":"https://benchlm.ai/benchmarks/graphwalksparents128k","paperUrl":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","year":"2026","fullName":"Graphwalks parents 0-128K","format":"Long-context graph reasoning","tasks":"Graph parent-retrieval tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"graphwalks-parents-<128k","url":"https://llm-stats.com/benchmarks/graphwalks-parents-<128k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false},{"catalog":"llm-stats","sourceId":"graphwalks-parents->128k","url":"https://llm-stats.com/benchmarks/graphwalks-parents->128k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","spatial reasoning","long context"],"catalogModelCount":11,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences","Long Context & Memory"],"domainScope":"general"},{"id":"bm_greekbarretrieval_75c379ac","familyId":"bmf_af9bde98b8ea","name":"GreekBarRetrieval","oneLine":"GreekBarRetrieval is a retrieval benchmark for Greek statutory articles, comprising bar-exam questions with case facts and candidate articles.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.IR"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-19","firstSeenAt":"2026-08-20","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.18752","pdf":"https://arxiv.org/pdf/2608.18752","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce GreekBarRetrieval, a public retrieval benchmark derived from, and complementing GreekBarBench, which did not include retrieval.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.18752"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GreekBarRetrieval is a retrieval benchmark for Greek statutory articles, comprising bar-exam questions with case facts and candidate articles.","whyItMatters":"Statutory retrieval for Greek is underexplored, and this benchmark provides a testbed for retrieval methods in a low-resource legal domain.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"685bc5dad810894c79f23dd3e4b5ff645c3966a7a3ada1db977041a2b8e650b5"},"motivation":"Statutory retrieval is necessary for citation-grounded legal question answering, but remains underexplored for Greek.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-20T09:45:58.504563Z","model":"deepseek-v4-flash","decisionReason":"Provides a public benchmark with defined data, protocol, and evaluation metrics; includes baseline experiments and is accessible via ArXiv. The paper does not specify an ongoing scoring service, but the dataset and protocol are reusable."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.18752","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-20T09:44:47.308583Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_grit-benchmark_0cf59789","familyId":"bmf_39727565f571","name":"GRIT","oneLine":"Evaluates LLM planners on robot skill execution under turbulence, scoring recovery from disruptions across tasks, tiers, and seeds.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-24","recognitionConfidence":0.4,"links":{"report":"https://github.com/guptabhishekumar/grit-benchmark","pdf":null,"project":"https://mujoco.org","code":"https://github.com/guptabhishekumar/grit-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"grit-benchmark GRIT: a benchmark scoring LLM planners that drive robot skills while the world undoes their work.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:guptabhishekumar/grit-benchmark"},"ranking":{"30d":{"score":23,"rank":121,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":325,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates LLM planners on robot skill execution under turbulence, scoring recovery from disruptions across tasks, tiers, and seeds.","whyItMatters":"Moves robot planning evaluation from clean execution to recovery under adversarial conditions, distinguishing robust planners from those that only verify.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"e11231de7e4841e9e486639d8b23905682df962dd3430327b64d1a99be027266"},"motivation":"grit-benchmark GRIT: a benchmark scoring LLM planners that drive robot skills while the world undoes their work.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/guptabhishekumar/grit-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-24T07:04:32.475961Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"Robot turbulence benchmarking is a specialized but emerging area, and the repository provides a complete harness and results, though frontier-model adoption is pending."},"evaluationMode":"public_reusable","publishers":[{"name":"GRIT authors","organizationType":"academic-lab","sourceUrl":"https://github.com/guptabhishekumar/grit-benchmark","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_groundtruth-dynamic-benchmarking_4f938e9e","familyId":"bmf_d6fd851014b2","name":"Groundtruth Dynamic Benchmarking","oneLine":"Geology question sets and grading rubrics for evaluating LLMs on real-world geological reasoning, with 153 questions across four configs, each grounded in source corpora with evidence locators and a harness for generation and scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://huggingface.co/datasets/EigenformAI/groundtruth-dynamic-benchmarking","pdf":null,"project":null,"code":null,"data":"https://huggingface.co/datasets/EigenformAI/groundtruth-dynamic-benchmarking","hfPaper":null},"evidence":{"snippet":"Groundtruth Dynamic Benchmarking — Geology Question sets and grading rubrics for evaluating LLMs on real-world geological reasoning.","reasonCodes":["discovered via huggingface","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":193,"hfDatasetLikes":0},"source":{"type":"huggingface","id":"huggingface:eigenformai/groundtruth-dynamic-benchmarking"},"ranking":{"30d":{"score":51,"rank":25,"coverage":0.15,"confidence":"Low","datasetDownloadRank":14,"datasetRankPopulation":30},"90d":{"score":48,"rank":81,"coverage":0.3,"confidence":"Low","datasetDownloadRank":38,"datasetRankPopulation":66}},"description":"Geology question sets and grading rubrics for evaluating LLMs on real-world geological reasoning, with 153 questions across four configs, each grounded in source corpora with evidence locators and a harness for generation and scoring.","whyItMatters":"Provides an evidence-grounded evaluation for domain-specific reasoning with rubric-based scoring, enabling reproducible model comparison on geology tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-26T06:20:35.651311Z","inputHash":"4c19ad07f3267ca6c1e39db5102561f70cfd4f51ae46ba053215c42d2c90e414"},"motivation":"Groundtruth Dynamic Benchmarking — Geology Question sets and grading rubrics for evaluating LLMs on real-world geological reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://huggingface.co/datasets/EigenformAI/groundtruth-dynamic-benchmarking","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":35,"confidence":"Low","horizon":"7d","reason":"Specialized domain benchmark may have limited but engaged interest from geology and LLM evaluation communities."},"evaluationMode":"score_submission","publishers":[{"name":"EigenformAI","organizationType":"community","sourceUrl":"https://huggingface.co/datasets/EigenformAI/groundtruth-dynamic-benchmarking","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_bc257a1b3101252a","familyId":"catalog_family_bc257a1b3101252a","name":"GroundUI-1K","oneLine":"A subset of GroundUI-18K for UI grounding evaluation, where models must predict action coordinates on screenshots based on single-step instructions across web, desktop, and mobile platforms.","description":"A subset of GroundUI-18K for UI grounding evaluation, where models must predict action coordinates on screenshots based on single-step instructions across web, desktop, and mobile platforms.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Grounding","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/groundui-1k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bc257a1b3101252a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/groundui-1k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"groundui-1k","url":"https://llm-stats.com/benchmarks/groundui-1k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","grounding","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_grouptom-bench_12e21c5f","familyId":"bmf_e0f69bf1e272","name":"GroupToM-Bench","oneLine":"GroupToM-Bench is a multimodal benchmark for evaluating group-level theory of mind through a seven-level cognitive audit framework.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04184","pdf":"https://arxiv.org/pdf/2606.04184","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04184"},"evidence":{"snippet":"We present GroupToM-Bench, the first multimodal benchmark for group-level ToM, built around a causal chain spanning micro-level BDI states (belief, desire, intention), meso-level group tension and structural constraints, and macro-level outcome prediction and mechanistic attribution.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04184"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GroupToM-Bench is a multimodal benchmark for evaluating group-level theory of mind through a seven-level cognitive audit framework.","whyItMatters":"It probes a gap in social cognition capabilities of multimodal LLMs, with implications for understanding emergent group behavior.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9680814c9d1989b4a3c300d553df134a3700805fa2927ed01bc46dfaacb3f9f7"},"motivation":"True general intelligence requires not only a model of the physical world but also a social world model: the capacity to infer how individual mental states interact and crystallize into group-level outcomes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04184","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_grouptravelbench_058d38b0","familyId":"bmf_014c2d359ad4","name":"GroupTravelBench","oneLine":"A benchmark for multi-user, multi-turn travel planning with 650 tasks and a synchronous group-chat sandbox, evaluating elicitation, coordination, and fairness-aware planning.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.25200","pdf":"https://arxiv.org/pdf/2605.25200","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.25200"},"evidence":{"snippet":"To bring the task back to its multi-user reality, we introduce \\textbf{\\textit{GroupTravelBench}}, the first benchmark for \\textbf{multi-user, multi-turn} travel planning.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25200"},"ranking":{},"description":"A benchmark for multi-user, multi-turn travel planning with 650 tasks and a synchronous group-chat sandbox, evaluating elicitation, coordination, and fairness-aware planning.","whyItMatters":"Highlights the challenge of group-level outcome quality for LLM agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"44f977987f1b71d55d5c2296f171786dc03390dd7f2adcbe05fa246cdbd9ed3d"},"motivation":"Travel planning in the real world is overwhelmingly a \\textit{group} activity, yet existing LLM travel-planning benchmarks reduce it to a single user, where the field is approaching saturation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25200","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e0d03a02444be465","familyId":"catalog_family_e0d03a02444be465","name":"GSM-8K (CoT)","oneLine":"Grade School Math 8K with Chain-of-Thought prompting, featuring 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.","description":"Grade School Math 8K with Chain-of-Thought prompting, featuring 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/gsm-8k-(cot)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e0d03a02444be465"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gsm-8k-(cot)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"gsm-8k-(cot)","url":"https://llm-stats.com/benchmarks/gsm-8k-(cot)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_0eb1f54b7f307458","familyId":"catalog_family_0eb1f54b7f307458","name":"GSM8K","oneLine":"Grade School Math 8K, a dataset of 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.","description":"Grade School Math 8K, a dataset of 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0eb1f54b7f307458"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/gsm8k"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gsm8k"}],"catalogSources":[{"catalog":"benchlm","sourceId":"gsm8k","url":"https://benchlm.ai/benchmarks/gsm8k","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"Grade School Math 8K","format":"Exact match","tasks":"Grade-school math word problems","successorKey":null},{"catalog":"llm-stats","sourceId":"gsm8k","url":"https://llm-stats.com/benchmarks/gsm8k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":48,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_dce7019d8b7b470d","familyId":"catalog_family_dce7019d8b7b470d","name":"GSM8K Chat","oneLine":"Grade School Math 8K adapted for chat format evaluation, featuring 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.","description":"Grade School Math 8K adapted for chat format evaluation, featuring 8.5K high-quality linguistically diverse grade school math word problems requiring multi-step reasoning and elementary arithmetic operations.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/gsm8k-chat","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dce7019d8b7b470d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/gsm8k-chat"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"gsm8k-chat","url":"https://llm-stats.com/benchmarks/gsm8k-chat","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_gst-bench_e194aeb7","familyId":"bmf_dc90c37b15cc","name":"GST-Bench","oneLine":"GST-Bench is a VQA benchmark for global spatial intelligence in video understanding, covering synthetic videos and human-verified questions. It evaluates VLMs' ability to infer spatial relations from novel viewpoints and map egocentric observations to top-down views.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.05747","pdf":"https://arxiv.org/pdf/2608.05747","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.05747"},"evidence":{"snippet":"To probe the cause of this gap, we construct GST-Bench-Local and find that models, despite strong local spatial understanding under the same task formulation, still fail to consolidate long-horizon observations into a globally consistent scene representation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":46,"hfDailySubmittedAt":"2026-08-07T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.05747"},"ranking":{"30d":{"score":55,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":53,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"GST-Bench is a VQA benchmark for global spatial intelligence in video understanding, covering synthetic videos and human-verified questions. It evaluates VLMs' ability to infer spatial relations from novel viewpoints and map egocentric observations to top-down views.","whyItMatters":"Existing video benchmarks focus on local spatial perception, while GST-Bench targets global spatial awareness over long-horizon videos, addressing a gap in evaluating embodied agents' spatial intelligence. It provides a scoring contract to compare VLMs and highlights a significant performance gap between models and humans.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"67d2fc1f118b7574e621f0d8db9a585f23e343d8f2dc5b26b08bd9c23b8272a4"},"motivation":"Spatial intelligence is fundamental to embodied agents, yet existing benchmarks focus on local spatial perception from single or few viewpoints, overlooking global spatial awareness over continuous, long-horizon visual streams.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.05747","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_gtbench_117f8151","familyId":"bmf_ed4666e9eda2","name":"GTBench","oneLine":"GTBench is a curriculum-grounded benchmark for evaluating LLMs as mathematical research assistants in graph theory, with problems across three difficulty groups.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.03144","pdf":"https://arxiv.org/pdf/2606.03144","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03144"},"evidence":{"snippet":"We introduce GTBench, a curriculum-grounded benchmark for evaluating LLMs as mathematical research assistants in graph theory, comprising 63 problems organized into three groups of increasing difficulty: undergraduate definitions and basic properties (Group 1), algorithm tracing and structural reasoning (Group 2), and graduate-level proof construction (Group 3).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03144"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"GTBench is a curriculum-grounded benchmark for evaluating LLMs as mathematical research assistants in graph theory, with problems across three difficulty groups.","whyItMatters":"It reveals performance gaps in graph-theoretic reasoning among frontier models, informing the use of AI in education and research.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3207b65649ae147fce81ae4e0f6d22067808f5d9aa62311cba32944cb66deccc"},"motivation":"Large language models (LLMs) are increasingly used as self-study assistants in technical disciplines, yet their reliability as mathematical reasoning assistants remains poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03144","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_guardian-crawler_0a6465c9","familyId":"bmf_bbf29cd625e1","name":"Guardian Crawler","oneLine":"Guardian Crawler is a retrieval-first testbed for knowledge discovery over synthetic web-like corpora, combining BM25 retrieval with reranking and constrained generation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval","Factuality"],"topics":["cs.IR"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.08994","pdf":"https://arxiv.org/pdf/2608.08994","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.08994"},"evidence":{"snippet":"We present Guardian Crawler, a reproducible retrieval-first testbed for controlled experiments on knowledge discovery and evidence-grounded summarization over synthetic web-like corpora.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08994"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Guardian Crawler is a retrieval-first testbed for knowledge discovery over synthetic web-like corpora, combining BM25 retrieval with reranking and constrained generation.","whyItMatters":"It provides a controlled environment for experiments on evidence-grounded summarization in noisy domains.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"56b51d3eb5ed6b35547e3a1ec438410d1ded469e82b2c9d748b8ac7e8b73c039"},"motivation":"Retrieving relevant evidence from noisy web data is challenging, particularly in sensitive domains containing incomplete reports, heterogeneous language, and irrelevant content.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"as a Short Paper at KDIR 2026 (International Conference on Knowledge Discovery and Information Retrieval)","evidence":"8 pages, 2 figures. Accepted as a Short Paper at KDIR 2026 (International Conference on Knowledge Discovery and Information Retrieval)","evidenceUrl":"https://arxiv.org/abs/2608.08994","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"as a Short Paper at KDIR 2026 (International Conference on Knowledge Discovery and Information Retrieval)","reviewStatus":"accepted","decisionRaw":"8 pages, 2 figures. Accepted as a Short Paper at KDIR 2026 (International Conference on Knowledge Discovery and Information Retrieval)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.08994","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"8 pages, 2 figures. Accepted as a Short Paper at KDIR 2026 (International Conference on Knowledge Discovery and Information Retrieval)","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_guardianagentbench_89ff40f2","familyId":"bmf_6f6ba219145f","name":"GuardianAgentBench","oneLine":"GuardianAgentBench (GABench) is a benchmark of 580 agent scenarios across six domains, evaluated on three production frameworks (LangChain, LlamaIndex, Vectara), with multi-stage validation and five adversarial attack modes to assess tool-use correctness and safety.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.20982","pdf":"https://arxiv.org/pdf/2607.20982","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20982"},"evidence":{"snippet":"We present GuardianAgentBench (GABench), a benchmark of 580 scenarios across six domains evaluated on three production-ready frameworks: LangChain, LlamaIndex, and Vectara.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20982"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"GuardianAgentBench (GABench) is a benchmark of 580 agent scenarios across six domains, evaluated on three production frameworks (LangChain, LlamaIndex, Vectara), with multi-stage validation and five adversarial attack modes to assess tool-use correctness and safety.","whyItMatters":"LLM agents need structured evaluation of tool selection and failure modes; GABench provides a reusable framework with multiple attack modes and guardrail assessment, offering a fine-grained view of where agents fail.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"969e7de6bcde2872f638de42b7d9814e7c538c18998495eee238c33023901d95"},"motivation":"As large language model agents increasingly operate autonomously with access to tools and external environments, ensuring their safe and reliable behavior becomes critical.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20982","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GuardianAgentBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.20982","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_guardianbench_debc3a47","familyId":"bmf_e7335f67edf2","name":"GuardianBench","oneLine":"3,024 instruction-scene examples organized as same-scene Safe/Unsafe contrastive pairs across hazard categories, based on international safety standards. Evaluates VLM safety reasoning under latent contextual risk.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-22","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.21928v1","pdf":"https://arxiv.org/pdf/2608.21928v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce GuardianBench, an instruction-contrastive benchmark grounded in international safety standards that isolates this latent contextual risk through 3,024 instruction-scene examples organized as same-scene Safe/Unsafe contrastive pairs across various hazard categories.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.21928"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"3,024 instruction-scene examples organized as same-scene Safe/Unsafe contrastive pairs across hazard categories, based on international safety standards. Evaluates VLM safety reasoning under latent contextual risk.","whyItMatters":"Exposes a critical failure mode in embodied AI safety—instruction-insensitive verdicts—and provides a controlled suite for improvement research.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-26T06:20:35.651311Z","inputHash":"89fe92b5cf418e1bfc88a593b72c69ccc4a097ba9118347e7a464353ac43dd24"},"motivation":"In embodied AI, safety risk can be latent: a benign instruction and a safe scene become hazardous only when composed.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The paper defines a benchmark with a clear task, scoring contract, and released suite, though direct download links are absent from supplied metadata.","canonicalNameSource":"paper_title","canonicalNameEvidence":"GuardianBench: A Same-Scene Instruction-Contrastive Benchmark for Latent Contextual Risk in Embodied AI"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.21928v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":50,"confidence":"Medium","horizon":"7d","reason":"Direct relevance to embodied AI safety may attract attention from safety-focused researchers."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_gui-primitives-diagnosing-spatial-reasonin_a22b1551","familyId":"bmf_29e2d419d7ef","name":"GUI-Primitives","oneLine":"994-item benchmark of contrastive instruction pairs over seven spatial relations in GUIs, isolating whether vision-language models bind relational language to UI elements correctly.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Safety","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-22","firstSeenAt":"2026-08-25","recognitionConfidence":0.6,"links":{"report":"http://arxiv.org/abs/2608.21832v1","pdf":"https://arxiv.org/pdf/2608.21832v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce GUI-Primitives, a 994-item benchmark of contrastive instruction pairs over seven spatial relations in graphical user interfaces (left/right, above/below, containment, alignment, proximity, list ordinal, occlusion).","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.21832"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"994-item benchmark of contrastive instruction pairs over seven spatial relations in GUIs, isolating whether vision-language models bind relational language to UI elements correctly.","whyItMatters":"Provides fine-grained diagnostics for spatial reasoning in GUI grounding, a key capability gap in computer-use agents, with public code and predictions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-26T06:20:35.651311Z","inputHash":"f035db2a2694c4a4cce7c744c092b1ab3d95d40440947a41bd70d5d53e869830"},"motivation":"Computer-use agents ground natural-language instructions in screenshots to locate interface elements, yet existing benchmarks do not isolate whether models bind relational language to the correct element.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The paper introduces a named benchmark with a controlled design, validated annotations, and released code and predictions.","canonicalNameSource":"paper_title","canonicalNameEvidence":"GUI-Primitives: Diagnosing Spatial Reasoning Failures in Vision-Language GUI Grounding"},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 Main Conference","evidence":"Accepted to EMNLP 2026 Main Conference","evidenceUrl":"http://arxiv.org/abs/2608.21832v1","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-25T17:37:25.889905Z"},"venueAttempts":[{"venueName":"EMNLP 2026 Main Conference","reviewStatus":"accepted","decisionRaw":"Accepted to EMNLP 2026 Main Conference","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"http://arxiv.org/abs/2608.21832v1","observedAt":"2026-08-25T17:37:25.889905Z","rawValue":"Accepted to EMNLP 2026 Main Conference","level":"author-claim"}]}],"attentionForecast":{"score":45,"confidence":"Low","horizon":"7d","reason":"Relevant to GUI agents and spatial reasoning, but scope may be narrower than general benchmarks."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_guideme_feb34852","familyId":"bmf_f4f8f3df85f0","name":"GuideMe","oneLine":"GuideMe is a benchmark for evaluating multimodal large language models on streaming video task guidance. It includes 2,458 videos (223.7 hours) with 47,775 interaction samples covering next-step instructions, completion feedback, error detection, and corrective guidance. Assessment uses temporal-semantic matching, behavioral classification, and LLM-as-a-Judge.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.02991","pdf":"https://arxiv.org/pdf/2607.02991","project":"https://fawnliu.github.io/project/guideme","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.02991"},"evidence":{"snippet":"In this paper, we construct GuideMe, the first multi-domain benchmark for streaming video that supports training and evaluation of MLLMs for closed-loop interactive task guidance.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02991"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"GuideMe is a benchmark for evaluating multimodal large language models on streaming video task guidance. It includes 2,458 videos (223.7 hours) with 47,775 interaction samples covering next-step instructions, completion feedback, error detection, and corrective guidance. Assessment uses temporal-semantic matching, behavioral classification, and LLM-as-a-Judge.","whyItMatters":"Existing multimodal models lack closed-loop interactive coaching ability; GuideMe provides a standardized evaluation for real-time procedural guidance, highlighting the gap in error detection and corrective feedback, which is crucial for practical assistants.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c6bf1e418be906dd2d54481190c0162801ab27f343cd9276d951faa6cda8e553"},"motivation":"While multimodal Large Language Models (MLLMs) excel at offline video understanding, an interesting question of how far they are from serving as a real-time procedural coach remains unknown.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02991","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GuideMe Project","organizationType":"academic-lab","sourceUrl":"https://fawnliu.github.io/project/guideme","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_guitestscape_3504136d","familyId":"bmf_ceab67585453","name":"GUITestScape","oneLine":"Evaluates exploratory GUI testing agents on 61 Android apps with 508 preset defects, using an open-set evaluator that decomposes trajectories into independently diagnosable capabilities.","area":"Agents & Tool Use","applicationDomains":["Consumer & Productivity"],"primaryDomain":"Consumer & Productivity","industrySectors":["Consumer Technology"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29532","pdf":"https://arxiv.org/pdf/2605.29532","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29532"},"evidence":{"snippet":"To address these challenges, we present GUITestScape, an interactive benchmark covering 61 real-world Android applications and 508 preset defects spanning interaction and display types, and introduce GUIJudge, an open-set evaluator that decomposes an agent's testing trajectory into independently diagnosable capabilities.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29532"},"ranking":{},"description":"Evaluates exploratory GUI testing agents on 61 Android apps with 508 preset defects, using an open-set evaluator that decomposes trajectories into independently diagnosable capabilities.","whyItMatters":"Addresses the lack of open-set evaluation in GUI testing, covering interaction and display defects. Provides a finer-grained assessment of agent capabilities and a verifier integration boost.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"ab255589bec9fea07fcf9736827fa3dcdddd5ccf8418b8d43d4dc5fb156e186e"},"motivation":"Exploratory GUI testing is a particularly demanding setting for MLLM agents: without predefined test scripts, an agent must autonomously navigate an application and discover defects through its own interaction.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29532","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GUITestScape Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2605.29532","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_h2hmem_6775c266","familyId":"bmf_5903d1e9207d","name":"H2HMem","oneLine":"H2HMem is a benchmark for evaluating memory capabilities of agents in human-human multimodal interactions, covering dyadic and multi-party conversations with tasks in memory recall, reasoning, and application.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09461","pdf":"https://arxiv.org/pdf/2606.09461","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.09461"},"evidence":{"snippet":"To address this gap, we introduce H2HMem, a Human-to-Human Multimodal Memory Benchmark for evaluating memory capabilities in complex human-human interactions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09461"},"ranking":{"90d":{"score":45,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"H2HMem is a benchmark for evaluating memory capabilities of agents in human-human multimodal interactions, covering dyadic and multi-party conversations with tasks in memory recall, reasoning, and application.","whyItMatters":"Existing memory benchmarks focus on single-user text interactions; H2HMem addresses the need for evaluating agents in complex multimodal human-human settings with asynchronous and conflicting information from multiple participants.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1dd0b93d91fc092466bc89b8b73897273985ddbd246c507c7d855c0c25cf004d"},"motivation":"Large language model agents are increasingly deployed in human-human interaction settings, such as meeting assistants and clinical documentation systems, where they must observe conversations and retain information for downstream queries.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09461","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_h2r-bench_da8120f6","familyId":"bmf_6a015e569fc5","name":"H2R-Bench","oneLine":"H2R-Bench evaluates cross-embodiment human-to-robot manipulation video generation. Models convert egocentric human demonstrations into robot manipulation videos under specified target embodiments. Scoring covers five dimensions: goal-state completion, action-event completion, functional contact transfer, embodiment correctness, and general video quality, aggregated into H2RCore.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.13049","pdf":"https://arxiv.org/pdf/2608.13049","project":null,"code":"https://github.com/Rongdingyi/H2R-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.13049"},"evidence":{"snippet":"Therefore, we introduce H2R-Bench, a benchmark for evaluating cross-embodiment human-to-robot manipulation video generation, where models transform egocentric human demonstrations into robot manipulation videos under specified embodiments.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":19,"hfDailySubmittedAt":"2026-08-14T00:00:00.000Z","githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13049"},"ranking":{"30d":{"score":35,"rank":61,"coverage":0.85,"confidence":"High"},"90d":{"score":33,"rank":209,"coverage":0.7,"confidence":"Medium"}},"description":"H2R-Bench evaluates cross-embodiment human-to-robot manipulation video generation. Models convert egocentric human demonstrations into robot manipulation videos under specified target embodiments. Scoring covers five dimensions: goal-state completion, action-event completion, functional contact transfer, embodiment correctness, and general video quality, aggregated into H2RCore.","whyItMatters":"Assesses whether video world models can bridge the embodiment gap between human hands and robotic end-effectors, providing a diagnostic for how well generated videos transfer functional interactions and task execution. Results show generic video quality does not correlate with transfer validity, aiding model selection for robot learning from human video.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ff0b215ab2661583ad04130b940ce1173170b802082fd527b01890a9a7ab5e0a"},"motivation":"Large-scale manipulation data is essential for robot learning, yet collecting robot demonstrations remains expensive and difficult to scale.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13049","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Shanghai Jiao Tong University","organizationType":"academic-lab","sourceUrl":"https://github.com/Rongdingyi/H2R-Bench","role":"benchmark-publisher"},{"name":"Shanghai Artificial Intelligence Laboratory","organizationType":"academic-lab","sourceUrl":"https://github.com/Rongdingyi/H2R-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_hack-verifiable-terminal-bench_a02835b2","familyId":"bmf_2c1e950a9953","name":"Hack-Verifiable Terminal Bench","oneLine":"Adapts Terminal Bench with hack-verifiable environments to automatically detect reward hacking in agent trajectories.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-22","firstSeenAt":"2026-08-25","recognitionConfidence":0.75,"links":{"report":"http://arxiv.org/abs/2608.22103v1","pdf":"https://arxiv.org/pdf/2608.22103v1","project":"https://majoroth.github.io/hack-verifiable-environments/hvtb","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"In this work, we adapt HVE to Terminal Bench, a leading benchmark of real-world terminal and coding tasks, and introduce Hack-Verifiable Terminal Bench (HVTB).","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22103"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Adapts Terminal Bench with hack-verifiable environments to automatically detect reward hacking in agent trajectories.","whyItMatters":"Provides reliable, automatic measurement of reward hacking, enabling comparison of model robustness without human judges.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"c495adf9a494f7eff1165afadc3827e976c42bbed5ac384c677bcd2497e1de9d"},"motivation":"As agents grow more capable and autonomous, their tendency to reward hack, satisfying a task's checks while violating its intent, becomes an increasingly important failure mode.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The paper explicitly releases environments and traces at a project page, providing a public reuse path and stable detection protocol.","canonicalNameSource":"paper_title","canonicalNameEvidence":"Hack-Verifiable Terminal Bench: Evaluating Reward Hacking in Terminal Tasks"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22103v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"The topic of reward hacking in autonomous agents is timely and the released environments support reproducibility."},"evaluationMode":"public_reusable","publishers":[{"name":"HVTB Team","organizationType":"academic-lab","sourceUrl":"https://majoroth.github.io/hack-verifiable-environments/hvtb","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_hakari-bench_2bd08ca4","familyId":"bmf_c03e69e4986c","name":"HAKARI-Bench","oneLine":"Evaluates retrieval architectures and efficiency settings (dimensionality reduction, quantization, reranking) across 35 benchmarks and 551 tasks in 43 languages, with unified conditions and metrics for BM25, dense, sparse, late interaction, and reranker models.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.IR"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22778","pdf":"https://arxiv.org/pdf/2606.22778","project":null,"code":"https://github.com/hakari-bench/hakari-bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.22778"},"evidence":{"snippet":"We present HAKARI-Bench, a lightweight benchmark that reconstructs existing retrieval suites into small datasets (Nano-sets): 35 benchmarks and 551 tasks across 43 languages in a unified format, enabling same-condition, model-agnostic comparison of five retrieval families (BM25, dense, sparse, late interaction, rerankers) and their efficiency variants.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":"2026-06-23T00:00:00.000Z","githubStars":30,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22778"},"ranking":{"90d":{"score":43,"rank":134,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates retrieval architectures and efficiency settings (dimensionality reduction, quantization, reranking) across 35 benchmarks and 551 tasks in 43 languages, with unified conditions and metrics for BM25, dense, sparse, late interaction, and reranker models.","whyItMatters":"Fills the gap for a lightweight, high-fidelity proxy for full retrieval benchmarks, enabling rapid model selection, regression detection, and quality-efficiency trade-off analysis under consistent conditions, which is otherwise computationally prohibitive.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1a479749837c506d9b8e8f26a60193dda51557e0f8f714b4c24bdf6473759177"},"motivation":"With the rapid spread of retrieval-augmented generation and semantic search, choosing the right embedding and retrieval configuration is increasingly hard.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22778","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_hakushobench_550d3cd1","familyId":"bmf_40c99bc08d37","name":"HakushoBench","oneLine":"HakushoBench is a Japanese chart and table VQA benchmark built from 33 governmental white papers, containing 2,053 images across over 10 image types with manually annotated QA pairs. It evaluates vision-language models on deep holistic understanding of charts and tables.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.01132","pdf":"https://arxiv.org/pdf/2606.01132","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01132"},"evidence":{"snippet":"As a first instantiation, we introduce HakushoBench, a challenging Japanese chart and table VQA benchmark built from 33 governmental white papers.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01132"},"ranking":{},"description":"HakushoBench is a Japanese chart and table VQA benchmark built from 33 governmental white papers, containing 2,053 images across over 10 image types with manually annotated QA pairs. It evaluates vision-language models on deep holistic understanding of charts and tables.","whyItMatters":"HakushoBench addresses the scarcity of non-English benchmarks for chart and table understanding, providing a challenging evaluation for VLMs in Japanese document domains. It reveals a notable performance gap between open-weight and proprietary models, highlighting areas for improvement in multilingual document AI.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"378f519992e7e3b1ad147dd8fbcef2ebcecc8c213ba809501addb3b6ae57375c"},"motivation":"Understanding chart and table images is essential for applying vision-language models (VLMs) to real-world document understanding.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01132","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HakushoBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.01132","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_8b46f9ccf66ae297","familyId":"catalog_family_8b46f9ccf66ae297","name":"Hallusion Bench","oneLine":"A comprehensive benchmark designed to evaluate image-context reasoning in large visual-language models (LVLMs) by challenging models with 346 images and 1,129 carefully crafted questions to assess language hallucination and visual illusion","description":"A comprehensive benchmark designed to evaluate image-context reasoning in large visual-language models (LVLMs) by challenging models with 346 images and 1,129 carefully crafted questions to assess language hallucination and visual illusion","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/hallusion-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8b46f9ccf66ae297"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/hallusion-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"hallusion-bench","url":"https://llm-stats.com/benchmarks/hallusion-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","vision"],"catalogModelCount":18,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_hallutruthqa_4b25f1be","familyId":"bmf_c7b1ef2118b6","name":"HalluTruthQA","oneLine":"HalluTruthQA is a fine-grained benchmark for Arabic QA hallucination detection, localization, and explanation, containing 2,400 expert-curated examples across four knowledge domains with span-level and explanation annotations.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20219","pdf":"https://arxiv.org/pdf/2607.20219","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20219"},"evidence":{"snippet":"We introduce HalluTruthQA, a fine-grained benchmark for hallucination evaluation in Arabic question answering.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20219"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"HalluTruthQA is a fine-grained benchmark for Arabic QA hallucination detection, localization, and explanation, containing 2,400 expert-curated examples across four knowledge domains with span-level and explanation annotations.","whyItMatters":"Hallucination evaluation typically uses response-level labels; this benchmark provides granular annotations to assess localization and explanation capabilities, but lacks a public reuse path for broader comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a8702053efd02a3f00deeb38ee306d006f272b7b6762eb650084998484884ac2"},"motivation":"Large language models (LLMs) can generate fluent Arabic answers, yet factual errors remain difficult to detect, localize, explain, and verify.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20219","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_hamqasbench_9db953d2","familyId":"bmf_9eb34c8bc18c","name":"HamQASBench","oneLine":"HamQASBench evaluates Quantum Architecture Search methods across 11 molecules organized in five structural tiers, using energy accuracy, per-qubit entanglement, and pairwise state fidelity to diagnose structural failures.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Quantum Technology"],"capabilities":[],"topics":["quant-ph"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.04845","pdf":"https://arxiv.org/pdf/2607.04845","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.04845"},"evidence":{"snippet":"We introduce HamQASBench, a Hamiltonian-informed diagnostic benchmark organizing 11 molecules into five structural tiers via fingerprints derived from the Pauli operator basis, computational basis representation, and ground-state entanglement.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.04845"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HamQASBench evaluates Quantum Architecture Search methods across 11 molecules organized in five structural tiers, using energy accuracy, per-qubit entanglement, and pairwise state fidelity to diagnose structural failures.","whyItMatters":"Existing QAS benchmarks miss structural failures like over-parameterization. This benchmark provides diagnostic metrics that reveal method-specific weaknesses, guiding algorithm development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"98fa095813bd8c2a2ba957fc87d0003925ab6a8e53ac13e1fcec04b21ae308df"},"motivation":"Quantum Architecture Search (QAS) automates the design of parameterized quantum circuits for variational quantum algorithms, yet existing benchmarks organize instances by molecular identity or qubit count -- criteria agnostic to Hamiltonian structure -- and rely solely on energy accuracy, which cannot detect structural failures such as over-parameterization on near-product ground states.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.04845","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_hardmtbench_3eaf8a99","familyId":"bmf_91528a6e04ed","name":"HardMTBench","oneLine":"HardMTBench is a difficulty-aware diagnostic benchmark for Chinese-English domain translation, covering 12 domains with 20,000 directional test items and annotated hardness scores based on domain knowledge, translation difficulty, and terminology load.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.28315","pdf":"https://arxiv.org/pdf/2605.28315","project":null,"code":"https://github.com/jasonNLP/HardMTBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.28315"},"evidence":{"snippet":"We introduce HardMTBench, a difficulty-aware diagnostic benchmark for bidirectional Chinese-English domain translation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28315"},"ranking":{},"description":"HardMTBench is a difficulty-aware diagnostic benchmark for Chinese-English domain translation, covering 12 domains with 20,000 directional test items and annotated hardness scores based on domain knowledge, translation difficulty, and terminology load.","whyItMatters":"Addresses the saturation of general MT benchmarks on Chinese-English by widening score separation, exposing domain-specific weaknesses in knowledge-intensive areas that quality-only metrics miss.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b9f7f5ff378782dbc14a56d78cc76f8200856ce363e51fbafcf0000647111757"},"motivation":"General-purpose machine translation benchmarks such as FLORES-200 have reached a saturation regime on Chinese-English pairs, where modern large language models cluster within a narrow band of high scores.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28315","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_harmurlbench_cbd88280","familyId":"bmf_2962ddda58cf","name":"HarmURLBench","oneLine":"AgentREVEAL is a diagnostic framework that evaluates safety alignment degradation in LLM agents when web retrieval is integrated. It assesses the impact of retrieval integration and content properties on harmful compliance, using a set of harmful behaviors and retrieval sources.","area":"Safety & Trustworthiness","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":["Information retrieval"],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29224","pdf":"https://arxiv.org/pdf/2605.29224","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29224"},"evidence":{"snippet":"We introduce HarmURLBench, a benchmark containing 1,405 real-world URLs paired with 320 harmful behaviors to support future evaluations.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29224"},"ranking":{},"description":"AgentREVEAL is a diagnostic framework that evaluates safety alignment degradation in LLM agents when web retrieval is integrated. It assesses the impact of retrieval integration and content properties on harmful compliance, using a set of harmful behaviors and retrieval sources.","whyItMatters":"AgentREVEAL addresses the underexplored risk that safety-aligned LLMs become more compliant with harmful requests when augmented with web retrieval. Its findings highlight a safety-utility trade-off that informs the design of safer retrieval-enabled agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"52ef7800702e2fbae6e4cb6a8150cc4b1cba78efbbe66c98874fe2e988564532"},"motivation":"AI agents augment large language models with external tools such as web retrieval, enabling grounded and up-to-date responses.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29224","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness","Search & Retrieval"],"domainScope":"specific"},{"id":"bm_harmvideobench_cc5648e1","familyId":"bmf_7d36f4d3eff5","name":"HarmVideoBench","oneLine":"Evaluates harmful video understanding in large multimodal models using 1,379 videos and 4,137 multiple-choice questions across three hierarchical dimensions: Observable Evidence, Clip-Internal Meaning, and Beyond-Clip Reasoning.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.27187","pdf":"https://arxiv.org/pdf/2606.27187","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.27187"},"evidence":{"snippet":"To address these problems, we present HarmVideoBench, a multi-layered diagnostic benchmark comprising 1,379 videos paired with 4,137 multiple-choice questions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.27187"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates harmful video understanding in large multimodal models using 1,379 videos and 4,137 multiple-choice questions across three hierarchical dimensions: Observable Evidence, Clip-Internal Meaning, and Beyond-Clip Reasoning.","whyItMatters":"Addresses the gap of shallow binary classification in harmful video benchmarks by testing deep contextual understanding, and the absence of explanatory rationales, providing a diagnostic tool for model evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"da19fcc6a36e1ba6b9e1ab2fdf885087bfa948216d6cc4da0be732640dff2ffe"},"motivation":"Large vision-language models (LVLMs) have recently shown immense potential in automated content moderation, sparking growing interest in developing harmful-video benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.27187","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_harness-bench_b6a94bb9","familyId":"bmf_5d1a38309c97","name":"Harness-Bench","oneLine":"Harness-Bench evaluates configuration-level harness effects in agent workflows with 106 sandboxed tasks, measuring completion, process quality, efficiency, and failure behavior across model-harness pairings.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27922","pdf":"https://arxiv.org/pdf/2605.27922","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27922"},"evidence":{"snippet":"We introduce Harness-Bench, a diagnostic benchmark for evaluating configuration-level harness effects in realistic agent workflows.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27922"},"ranking":{},"description":"Harness-Bench evaluates configuration-level harness effects in agent workflows with 106 sandboxed tasks, measuring completion, process quality, efficiency, and failure behavior across model-harness pairings.","whyItMatters":"Addresses a gap in agent evaluation by isolating harness configuration effects, showing that agent capability is configuration-level rather than model-only, and identifying execution-alignment failures.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9ebfc8b12156538dfa2d5e20681e9fc36cd36c0044c08f436d2d6be2156ac499"},"motivation":"LLM agents are increasingly deployed as executable systems that use tools, modify workspaces, and produce concrete artifacts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27922","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_harnessopt-bench_24067463","familyId":"bmf_42c79c547b84","name":"HarnessOpt-Bench","oneLine":"Evaluates LLMs optimizing a target agent's harness (prompts, tools, control flow) under budgeted, stochastic evaluation. Scoring is normalized gain over seed on a held-out test partition, with trusted execution environment enforcing evaluation boundary.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.06301","pdf":"https://arxiv.org/pdf/2608.06301","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.06301"},"evidence":{"snippet":"We introduce HarnessOpt-Bench, a benchmark for end-to-end harness optimization under expensive and stochastic evaluation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":35,"hfDailySubmittedAt":"2026-08-07T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06301"},"ranking":{"30d":{"score":54,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":52,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates LLMs optimizing a target agent's harness (prompts, tools, control flow) under budgeted, stochastic evaluation. Scoring is normalized gain over seed on a held-out test partition, with trusted execution environment enforcing evaluation boundary.","whyItMatters":"Addresses the gap in measuring automated harness optimization, a discriminative capability needed for improving agentic LLM systems. Provides a protocol for comparing optimizer models under varying harnesses and tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cb38c3cb1edd81673d8340dd632183705f15e21f4f19da9f33d5572fb4765412"},"motivation":"As LLMs are increasingly deployed within agentic systems, their capabilities depend not only on the model weights but also on the harness: the prompts, tools, control flow, memory, and orchestration code surrounding them.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06301","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_harnessrisk_e838b2b9","familyId":"bmf_171dcd6194e0","name":"HarnessRisk","oneLine":"Evaluates safety failures of model–agent-harness configurations across setup, runtime, persistent state, actions and incident recovery.","area":"Safety & Trustworthiness","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":[],"capabilities":["Agent safety","Prompt-injection resistance","Tool-use safety","Persistent-state safety"],"topics":["Agents","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.17597","pdf":"https://arxiv.org/pdf/2608.17597","project":"https://baiyajing.github.io/harness-risk/","code":"https://github.com/Baiyajing/HarnessRisk","data":"https://huggingface.co/datasets/YajingB/HarnessRisk","hfPaper":"https://huggingface.co/papers/2608.17597"},"evidence":{"snippet":"We present HarnessRisk, a lifecycle oriented benchmark that organizes agent harness safety into six operational phases including Harness Configuration, Capability Extension, Runtime Operation, State Persistence, Action Control, and Incident Recovery.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":10,"hfDailySubmittedAt":"2026-08-19T00:00:00.000Z","githubStars":11,"githubScope":"benchmark_repo","hfDatasetDownloads":62,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2608.17597"},"ranking":{"30d":{"score":40,"rank":49,"coverage":1.0,"confidence":"High","datasetDownloadRank":28,"datasetRankPopulation":30},"90d":{"score":35,"rank":195,"coverage":1.0,"confidence":"High","datasetDownloadRank":57,"datasetRankPopulation":66}},"description":"HarnessRisk evaluates agent harness safety across six lifecycle phases with 128 sandboxed cases. It measures Utility, Attack Success Rate, Persistence, and Detection for each trajectory.","whyItMatters":"Addresses the need for systematic evaluation of agent harness safety across multiple responsibilities, enabling comparison of harness and model combinations and highlighting vulnerability patterns.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fb1a3bd4756a474cb52f9101f8ae4686adcf1d399910348746b8097fbc44dac0"},"motivation":"Large language models are increasingly deployed through agent harnesses that manage tools, extensions, persistent state, permissions, and external actions.","constructionDetail":"HarnessRisk evaluates the safety of complete model–harness configurations across six operational lifecycle phases in fresh sandboxes.","detail":{"taskBreakdown":["Harness configuration","Capability extension","Runtime operation","State persistence","Action control","Incident recovery"],"protocol":{"tasks":"128 sandboxed cases across six lifecycle phases","primaryMetric":"Utility, Attack Success Rate, Persistence and Detection","version":"arXiv v1"}},"curation":{"state":"source-reviewed","reviewedAt":"2026-08-20","sources":["https://arxiv.org/abs/2608.17597","https://arxiv.org/html/2608.17597","https://github.com/Baiyajing/HarnessRisk","https://huggingface.co/datasets/YajingB/HarnessRisk"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17597","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"score_submission","availability":{"paperStatus":"available","githubStatus":"available","hfDatasetStatus":"available","evaluatorStatus":"available","submissionStatus":"not_found"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_harnesssafe_835e5e2c","familyId":"bmf_bde338e32b63","name":"HarnessSafe","oneLine":"HarnessSafe evaluates safety of agent harnesses through 328 executable cases across seven persistent-carrier families, using trace-based evaluation to track attack chains from entry to potential violation.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.06984","pdf":"https://arxiv.org/pdf/2608.06984","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.06984"},"evidence":{"snippet":"To this end, we present HarnessSafe, a benchmark comprising 328 executable cases across seven persistent-carrier families and evaluated on most mainstream agent harnesses.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06984"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HarnessSafe evaluates safety of agent harnesses through 328 executable cases across seven persistent-carrier families, using trace-based evaluation to track attack chains from entry to potential violation.","whyItMatters":"Addresses the lack of benchmarks covering multiple persistent carriers and providing trace-level analysis, offering a more nuanced assessment of safety risks in agent systems compared to end-to-end success rates.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9625ca8c248cc9847bc16bbed502a8d2ee8162f6614bd9356d9aa38c78b00e4f"},"motivation":"Modern agent harnesses persist state across tasks and sessions through persistent carriers like memory, skills, tools, and shared artifacts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06984","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_45bbbc773f376558","familyId":"catalog_family_45bbbc773f376558","name":"Harvey LAB (Vals)","oneLine":"Harvey LAB (Vals) is a professional legal-work evaluation of AI systems on complex law-firm style tasks, reported by Vals.","description":"Harvey LAB (Vals) is a professional legal-work evaluation of AI systems on complex law-firm style tasks, reported by Vals.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/harvey-lab","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_45bbbc773f376558"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/harvey-lab"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"harvey-lab","url":"https://llm-stats.com/benchmarks/harvey-lab","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_09556350c5c8a0a0","familyId":"catalog_family_09556350c5c8a0a0","name":"Harvey LAB-AA","oneLine":"Harvey LAB-AA evaluates model performance on complex legal workflows.","description":"Harvey LAB-AA evaluates model performance on complex legal workflows.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/harvey-lab-aa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_09556350c5c8a0a0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/harvey-lab-aa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"harvey-lab-aa","url":"https://llm-stats.com/benchmarks/harvey-lab-aa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_fe85ad67f7029338","familyId":"catalog_family_fe85ad67f7029338","name":"Harvey's Legal Agent Benchmark","oneLine":"Tests an agent's ability to complete legal work using documents, spreadsheets, presentations, and file-system tools","description":"Tests an agent's ability to complete legal work using documents, spreadsheets, presentations, and file-system tools","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/hlab","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fe85ad67f7029338"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/hlab"}],"catalogSources":[{"catalog":"benchlm","sourceId":"hlab","url":"https://benchlm.ai/benchmarks/hlab","paperUrl":"https://www.vals.ai/benchmarks/hlab","year":"2026","fullName":"Vals Harvey's Legal Agent Benchmark","format":"Accuracy score","tasks":"Legal agent work across documents, spreadsheets, presentations, and files","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_hawkesnest_c85cfbd3","familyId":"bmf_8ce9fda1f146","name":"HawkesNest","oneLine":"HawkesNest is a synthetic benchmark for spatiotemporal point process models, providing controlled generators and complexity ladders across four axes: space-time entanglement, background heterogeneity, cross-type interaction, and domain topology. It includes simulation, export, and visualization tools.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.16863","pdf":"https://arxiv.org/pdf/2606.16863","project":null,"code":"https://github.com/YahyaAalaila/HawkesNest","data":null,"hfPaper":"https://huggingface.co/papers/2606.16863"},"evidence":{"snippet":"We introduce HawkesNest, a generator-aligned benchmark for controlled spatiotemporal pattern complexity built on a multivariate Hawkes backbone.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16863"},"ranking":{"90d":{"score":36,"rank":187,"coverage":0.55,"confidence":"Low"}},"description":"HawkesNest is a synthetic benchmark for spatiotemporal point process models, providing controlled generators and complexity ladders across four axes: space-time entanglement, background heterogeneity, cross-type interaction, and domain topology. It includes simulation, export, and visualization tools.","whyItMatters":"Real-world spatiotemporal event datasets obscure generative structure, making model failures hard to attribute. HawkesNest isolates complexity factors for diagnostic stress tests, enabling controlled evaluation of STPP models under known structural difficulty.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"105ff62c8a3491a08c68a64cf21d1e3522cc2d33183682e2806a5d0fc4231176"},"motivation":"Evaluation of spatiotemporal point process (STPP) models relies heavily on opaque real-world datasets, where latent generative structure is unknown and model failures are difficult to attribute.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.16863","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Yahya Aalaila et al.","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.16863","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_healmed_cb6026bf","familyId":"bmf_a1a70ab23feb","name":"HealMed","oneLine":"Evaluates multilingual medical LLMs on 1,000 examples per language across nine languages, covering MCQA, NLI, and open-ended QA tasks.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.19981","pdf":"https://arxiv.org/pdf/2608.19981","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present HealMed, an expert-reviewed benchmark for multilingual evaluation of large language models in medicine.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.19981"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates multilingual medical LLMs on 1,000 examples per language across nine languages, covering MCQA, NLI, and open-ended QA tasks.","whyItMatters":"Fills a gap in multilingual medical evaluation by measuring performance across language resource levels and assessing whether translation quality affects cross-language results.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"cc5941b7d49014d2f308c1dad339087f544c4bdf316cecab5d88b29b8fffad4f"},"motivation":"We present HealMed, an expert-reviewed benchmark for multilingual evaluation of large language models in medicine.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"The paper explicitly presents HealMed as an expert-reviewed benchmark with defined task formats and a public arxiv report, satisfying stable scoring and a public reuse path.","canonicalNameSource":"paper_title","canonicalNameEvidence":"HealMed: Multilingual Evaluation of Large Language Models in Medicine"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.19981","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:50.392548Z"},"attentionForecast":{"score":62,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses a clear multilingual medical evaluation gap, but the arxiv-only release without explicit code or data links may limit immediate adoption."},"evaluationMode":"score_submission","publishers":[{"name":"Medical Experts and Physicians Across Nine Countries","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.19981","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_healthagentbench_400912dd","familyId":"bmf_38ba71a1d0f7","name":"HealthAgentBench","oneLine":"Evaluates AI agents on 54 realistic healthcare tasks across 7 categories, including medical imaging, EHR analysis, and clinical trial matching. Agents operate in terminal environments with task-specific verifiers and a final task success rate.","area":"Agents & Tool Use","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.31179","pdf":"https://arxiv.org/pdf/2606.31179","project":null,"code":"https://github.com/microsoft/HealthAgentBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.31179"},"evidence":{"snippet":"We introduce HealthAgentBench, a suite of 54 agentic healthcare tasks across 7 categories each with its unique environment.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":"2026-07-02T00:00:00.000Z","githubStars":44,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31179"},"ranking":{"90d":{"score":49,"rank":66,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates AI agents on 54 realistic healthcare tasks across 7 categories, including medical imaging, EHR analysis, and clinical trial matching. Agents operate in terminal environments with task-specific verifiers and a final task success rate.","whyItMatters":"Provides a standardized, realistic evaluation for agentic healthcare AI, revealing performance gaps across task types and informing model selection for clinical applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"65afc00a2c6c32b161975e9b50e9fdf9dcd8514bb9fac5e2b1ae58b65a1b4129"},"motivation":"As AI agents become increasingly capable of complex, long-horizon reasoning, rigorous and holistic evaluation is essential for measuring progress toward real-world healthcare applications.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31179","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Microsoft","organizationType":"company-research-lab","sourceUrl":"https://github.com/microsoft/HealthAgentBench","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"lib_healthbench","familyId":"family_healthbench","name":"HealthBench","oneLine":"Established benchmark family · Medical Reasoning.","area":"Medical Reasoning","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Medical Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2025-01-01","releaseDatePrecision":"year","firstRelease":{"year":2025,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2505.08775","pdf":null,"project":"https://openai.com/index/healthbench/","code":"https://github.com/openai/healthbench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_healthbench"},"ranking":{},"recordType":"family","aliases":[],"sourceAttribution":[{"role":"official-release","url":"https://openai.com/index/healthbench/"}],"adoptionRefs":["openai-gpt5"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific","catalogSources":[{"catalog":"llm-stats","sourceId":"healthbench","url":"https://llm-stats.com/benchmarks/healthbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["healthcare"],"catalogModelCount":9,"catalogStarCount":0},{"id":"catalog_fd69c6fcc3dcced8","familyId":"catalog_family_fd69c6fcc3dcced8","name":"HealthBench (length-adjusted)","oneLine":"HealthBench score after applying a verbosity penalty to model responses.","description":"HealthBench score after applying a verbosity penalty to model responses.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fd69c6fcc3dcced8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/healthbenchlengthadjusted"}],"catalogSources":[{"catalog":"benchlm","sourceId":"healthBenchLengthAdjusted","url":"https://benchlm.ai/benchmarks/healthbenchlengthadjusted","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"HealthBench length-adjusted score","format":"Length-adjusted rubric score","tasks":"5,000 multi-turn patient conversations","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_07dc52abdc9c64dc","familyId":"catalog_family_07dc52abdc9c64dc","name":"HealthBench (raw)","oneLine":"Raw score on realistic multi-turn healthcare conversations graded against expert-written rubrics.","description":"Raw score on realistic multi-turn healthcare conversations graded against expert-written rubrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_07dc52abdc9c64dc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/healthbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"healthBench","url":"https://benchlm.ai/benchmarks/healthbench","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"HealthBench raw score","format":"Raw rubric score","tasks":"5,000 multi-turn patient conversations","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_bd8fc219b7dfc4b0","familyId":"catalog_family_bd8fc219b7dfc4b0","name":"HealthBench Consensus","oneLine":"HealthBench Consensus is a HealthBench subset focused on questions where physician-created rubric criteria have especially high agreement, measuring healthcare performance and safety on consensus-evaluable conversations.","description":"HealthBench Consensus is a HealthBench subset focused on questions where physician-created rubric criteria have especially high agreement, measuring healthcare performance and safety on consensus-evaluable conversations.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/healthbench-consensus","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bd8fc219b7dfc4b0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/healthbench-consensus"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"healthbench-consensus","url":"https://llm-stats.com/benchmarks/healthbench-consensus","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["healthcare"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"lib_healthbench_hard","familyId":"family_healthbench","name":"HealthBench Hard","oneLine":"Established benchmark variant · Medical Reasoning.","area":"Medical Reasoning","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Medical Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2025-01-01","releaseDatePrecision":"year","firstRelease":{"year":2025,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2505.08775","pdf":null,"project":"https://openai.com/index/healthbench/","code":"https://github.com/openai/healthbench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_healthbench_hard"},"ranking":{},"recordType":"variant","aliases":["HealthBench-Hard"],"sourceAttribution":[{"role":"benchmark-definition","url":"https://openai.com/index/healthbench/"}],"adoptionRefs":["openai-gpt5"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"variantOf":"lib_healthbench","capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific","catalogSources":[{"catalog":"benchlm","sourceId":"healthBenchHard","url":"https://benchlm.ai/benchmarks/healthbench-hard","paperUrl":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","year":"2026","fullName":"HealthBench Hard","format":"Open-ended health evaluation","tasks":"1,000 health prompts","successorKey":null},{"catalog":"llm-stats","sourceId":"healthbench-hard","url":"https://llm-stats.com/benchmarks/healthbench-hard","datasetSlug":"healthbench","versionCount":4,"subsetCount":3,"rowCount":null,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":["knowledge","healthcare"],"catalogModelCount":9,"catalogStarCount":0},{"id":"lib_healthbench_professional","familyId":"family_healthbench","name":"HealthBench Professional","oneLine":"Established benchmark variant · Medical Reasoning.","area":"Medical Reasoning","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Medical Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2025-01-01","releaseDatePrecision":"year","firstRelease":{"year":2025,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2505.08775","pdf":null,"project":"https://openai.com/index/healthbench/","code":"https://github.com/openai/healthbench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_healthbench_professional"},"ranking":{},"recordType":"variant","aliases":["HealthBench-Professional"],"sourceAttribution":[{"role":"benchmark-definition","url":"https://openai.com/index/healthbench/"}],"adoptionRefs":["openai-gpt5"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"variantOf":"lib_healthbench","capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific","catalogSources":[{"catalog":"benchlm","sourceId":"healthBenchProfessional","url":"https://benchlm.ai/benchmarks/healthbenchprofessional","paperUrl":"https://arxiv.org/abs/2604.27470","year":"2026","fullName":"HealthBench Professional","format":"Rubric-graded open-ended responses","tasks":"Clinician chat tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"healthbench-professional","url":"https://llm-stats.com/benchmarks/healthbench-professional","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","healthcare"],"catalogModelCount":9,"catalogStarCount":0},{"id":"catalog_90d0d37829c16794","familyId":"catalog_family_90d0d37829c16794","name":"HealthBench Professional (raw)","oneLine":"Raw score on physician-authored clinical consult, documentation, and research conversations.","description":"Raw score on physician-authored clinical consult, documentation, and research conversations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_90d0d37829c16794"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/healthbenchprofessionalraw"}],"catalogSources":[{"catalog":"benchlm","sourceId":"healthBenchProfessionalRaw","url":"https://benchlm.ai/benchmarks/healthbenchprofessionalraw","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"HealthBench Professional raw score","format":"Raw rubric score","tasks":"525 physician-authored conversations","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_healthbench-psych_83387771","familyId":"bmf_d81079d8455c","name":"HealthBench-Psych","oneLine":"Evaluates LLMs on 610 mental-health conversations from HealthBench using a three-judge panel and physician rubrics, with a harder 119-conversation subset.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-25","firstSeenAt":"2026-08-27","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.25071","pdf":"https://arxiv.org/pdf/2608.25071","project":null,"code":"https://github.com/mindbench-ai/healthbench-psych","data":"https://huggingface.co/datasets/mindbench-ai/healthbench-psych","hfPaper":null},"evidence":{"snippet":"We introduce HealthBench-Psych and HealthBench-Psych-Hard.","reasonCodes":["exact named benchmark artifact released in abstract","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":82,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2608.25071"},"ranking":{"30d":{"score":7,"rank":168,"coverage":1.0,"confidence":"High","datasetDownloadRank":21,"datasetRankPopulation":30},"90d":{"score":11,"rank":408,"coverage":1.0,"confidence":"High","datasetDownloadRank":48,"datasetRankPopulation":66}},"description":"Evaluates LLMs on 610 mental-health conversations from HealthBench using a three-judge panel and physician rubrics, with a harder 119-conversation subset.","whyItMatters":"Provides a clinically validated mental-health evaluation subset with released data, code, and judge panel, enabling reproducible specialty-specific model assessment.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"9555af4373b5ee095401d81578f3d1ec04604f7487c9c76347e2ce2200884efb"},"motivation":"General-purpose health benchmarks increasingly anchor claims about LLM medical performance, but they are not always resolved by clinical specialty, making domain-specific performance hard to isolate.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"Includes released subset, pipeline, model responses, grades, and analysis code; offers a stable scoring protocol and public reuse path despite requiring access to the original HealthBench corpus.","canonicalNameSource":"paper_title","canonicalNameEvidence":"HealthBench-Psych: A Mental Health Subset of OpenAI's HealthBench"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25071","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"Mental-health LLM evaluation is a high-interest niche, and the released artifacts and evidence-backed rigor may drive moderate early attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_heart-bench_230e27fe","familyId":"bmf_1835f35879b7","name":"HEART-Bench","oneLine":"HEART-Bench evaluates whether LLM agents can simulate coherent, human-like psychology. It provides 11 fictional characters with raw episodic memories, 64 decision-making scenarios based on the DIAMONDS taxonomy, and 673 multiple-choice questions with expert-annotated ground truth. Two evaluation tracks: MCQ and open-ended consciousness-narrative.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30058","pdf":"https://arxiv.org/pdf/2605.30058","project":null,"code":"https://github.com/peng-weihan/HEART-BENCH","data":null,"hfPaper":"https://huggingface.co/papers/2605.30058"},"evidence":{"snippet":"In this paper, we introduce a novel benchmark to systematically assess whether LLM agents can simulate coherent, human-like psychology.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30058"},"ranking":{},"description":"HEART-Bench evaluates whether LLM agents can simulate coherent, human-like psychology. It provides 11 fictional characters with raw episodic memories, 64 decision-making scenarios based on the DIAMONDS taxonomy, and 673 multiple-choice questions with expert-annotated ground truth. Two evaluation tracks: MCQ and open-ended consciousness-narrative.","whyItMatters":"This benchmark addresses the gap in evaluating emotional and personality consistency in LLM agents, complementing task-oriented benchmarks. It offers a standardized protocol for assessing psychological coherence, which is valuable for developing agents that can maintain stable personas and make value-consistent decisions in interactive applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"60ecad22cf0bf9918ea5ec0cf1b21cea1621810990694d503c3a0440560ee793"},"motivation":"While LLM agents have demonstrated remarkable task-oriented abilities such as planning, reasoning, and action, few works have treated them as complete human personalities where emotional dimensions hold equal importance.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30058","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HEART-Bench team","organizationType":"academic-lab","sourceUrl":"https://github.com/peng-weihan/HEART-BENCH","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_heatcast_ee7ee561","familyId":"bmf_e7e61f16dc25","name":"HeatCast","oneLine":"HeatCast is a benchmark for monthly Land Surface Temperature forecasting at 30m resolution across 124 U.S. cities. It provides Landsat-based tiles, fixed temporal split, LCZ-stratified metrics, and a reference evaluation harness.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07640","pdf":"https://arxiv.org/pdf/2608.07640","project":"https://doi.org/10.57967/hf/9889","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07640"},"evidence":{"snippet":"We introduce HeatCast, a Landsat-based benchmark for monthly LST forecasting across 124 U.S.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07640"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HeatCast is a benchmark for monthly Land Surface Temperature forecasting at 30m resolution across 124 U.S. cities. It provides Landsat-based tiles, fixed temporal split, LCZ-stratified metrics, and a reference evaluation harness.","whyItMatters":"The benchmark addresses a gap in neighborhood-scale LST forecasting, offering a large-scale dataset with standardized metrics to compare forecasting models across diverse urban environments, supporting practical applications in urban heat monitoring.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3bba44789bcc04b64d3d0eb0e83bd9abe163c16015e710f969ee0fda9c0363e0"},"motivation":"Land Surface Temperature (LST) is a widely used satellite-derived measure of urban surface heat, but there is no shared benchmark for forecasting it at 30 m.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07640","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_hedge-bench_8e798fcd","familyId":"bmf_9e742fdac1fd","name":"Hedge-Bench","oneLine":"Hedge-Bench 1.0 evaluates agents on 102 financial reasoning tasks with deterministic grading against expert reasoning traces.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03918","pdf":"https://arxiv.org/pdf/2606.03918","project":null,"code":"https://github.com/Trata-Inc/trata-hedge-bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.03918"},"evidence":{"snippet":"We present Hedge-Bench 1.0: a benchmark of 102 actual, on-the-job tasks grounded in the explicit reasoning traces of professional hedge fund analysts working with relevant information sources.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":98,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03918"},"ranking":{"90d":{"score":52,"rank":51,"coverage":0.7,"confidence":"Medium"}},"description":"Hedge-Bench 1.0 evaluates agents on 102 financial reasoning tasks with deterministic grading against expert reasoning traces.","whyItMatters":"It provides a reproducible benchmark for open-ended financial analysis, addressing the lack of hard, realistic tasks with verified grading.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"244e330fede2224ae02d72cf1808c09c427087d1524f93275d02bf8d436e8111"},"motivation":"AI agents can increasingly handle the mechanical tasks of financial analysis: retrieving documents, calculating formulas, updating spreadsheets.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03918","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Trata Inc.","organizationType":"company-research-lab","sourceUrl":"https://github.com/Trata-Inc/trata-hedge-bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_hedgehog_997bad40","familyId":"bmf_064a43832b4f","name":"HEDGEHOG","oneLine":"HEDGEHOG is a six-stage filtration benchmark for evaluating molecular generators on medicinal plausibility, including physicochemical, structural, synthesis, docking, and 3D pose checks, applied to 23 generators and 230,000 molecules.","area":"Language & Knowledge","applicationDomains":["Science & Research","Health & Life Sciences"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals","Pharma & Biotech"],"capabilities":[],"topics":["cs.LG"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.13155","pdf":"https://arxiv.org/pdf/2607.13155","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.13155"},"evidence":{"snippet":"We introduce HEDGEHOG, a unified six-stage filtration benchmark that is inspired by industrial hit identification workflows: (i) preprocessing; (ii) physicochemical descriptor screening; (iii) structural alerts and graph-sanity checks; (iv) synthesis feasibility; (v) docking and binding affinity estimation; and (vi) three-dimensional pose and interaction checks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.13155"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HEDGEHOG is a six-stage filtration benchmark for evaluating molecular generators on medicinal plausibility, including physicochemical, structural, synthesis, docking, and 3D pose checks, applied to 23 generators and 230,000 molecules.","whyItMatters":"This benchmark exposes that current molecular generators fail to produce compounds that pass all medicinal, synthesis, docking, and 3D filters simultaneously, providing a more realistic evaluation for drug discovery.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ab4bed6f32fa784628a36408bc6ab0132c84c7ce1f1c60b6b1f2075ce176dbcb"},"motivation":"Generative molecular models can support early drug discovery by proposing new candidate compounds de novo.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.13155","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"catalog_99665bad41328c17","familyId":"catalog_family_99665bad41328c17","name":"HellaSwag","oneLine":"A challenging commonsense natural language inference dataset that uses Adversarial Filtering to create questions trivial for humans (>95% accuracy) but difficult for state-of-the-art models, requiring completion of sentence endings based on physical situations and everyday activities","description":"A challenging commonsense natural language inference dataset that uses Adversarial Filtering to create questions trivial for humans (>95% accuracy) but difficult for state-of-the-art models, requiring completion of sentence endings based on physical situations and everyday activities","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_99665bad41328c17"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/hellaswag"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/hellaswag"}],"catalogSources":[{"catalog":"benchlm","sourceId":"hellaswag","url":"https://benchlm.ai/benchmarks/hellaswag","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"HellaSwag","format":"Exact match","tasks":"Commonsense completion questions","successorKey":null},{"catalog":"llm-stats","sourceId":"hellaswag","url":"https://llm-stats.com/benchmarks/hellaswag","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":27,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_herabench_8a4ad4a5","familyId":"bmf_c80a435ff4de","name":"HeraBench","oneLine":"HeraBench is a fault-injected benchmark for multi-device agent workflows on Linux and Android, evaluating hierarchical replanning under injected strategy- and device-level failures.","area":"Agents & Tool Use","applicationDomains":["Consumer & Productivity"],"primaryDomain":"Consumer & Productivity","industrySectors":["Consumer Technology"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.20487","pdf":"https://arxiv.org/pdf/2606.20487","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.20487"},"evidence":{"snippet":"To evaluate this capability, we introduce \\textbf{HeraBench}, a fault-injected benchmark that constructs cross-device workflows over Linux and Android devices and injects strategy- and device-level failures.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.20487"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"HeraBench is a fault-injected benchmark for multi-device agent workflows on Linux and Android, evaluating hierarchical replanning under injected strategy- and device-level failures.","whyItMatters":"Current multi-device agent benchmarks lack systematic fault injection to test recovery capabilities, making it difficult to compare hierarchical versus global replanning approaches.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"33d4b5b3d23fa2335ee9396a6fb77b16e6957b669a600505ea0b05387a0f4197"},"motivation":"Real-world computer-use tasks often span multiple applications and devices, requiring agents to coordinate heterogeneous environments under dynamic runtime failures.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.20487","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_hero-s-journey_260791d3","familyId":"bmf_c5187f03b592","name":"HERO'S JOURNEY","oneLine":"Tests rule induction in goal-directed episodic tasks through text games, covering eight tasks across attribute and procedural induction families with controllable lexical grounding and identifiability conditions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02556","pdf":"https://arxiv.org/pdf/2606.02556","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02556"},"evidence":{"snippet":"We introduce HERO'S JOURNEY, a benchmark for rule induction in goal-directed episodic tasks, where agents must infer hidden rules from demonstrations and act on them through multi-step execution.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02556"},"ranking":{},"description":"Tests rule induction in goal-directed episodic tasks through text games, covering eight tasks across attribute and procedural induction families with controllable lexical grounding and identifiability conditions.","whyItMatters":"Evaluates a specific cognitive capability in LLMs, revealing limitations in procedural induction and execution bottlenecks. Provides a structured testbed for studying rule learning and guiding method development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"f2300d37789974d7bc2d7c042e56275c010d65a35c8714a823251140a5f065fa"},"motivation":"We introduce HERO'S JOURNEY, a benchmark for rule induction in goal-directed episodic tasks, where agents must infer hidden rules from demonstrations and act on them through multi-step execution.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02556","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HERO'S JOURNEY Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.02556","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_hg-bench_1e72857e","familyId":"bmf_5e2cc69b8f13","name":"HG-Bench","oneLine":"HG-Bench evaluates page-aware, two-level answer-region grounding: given multi-page handwritten homework images, models must localize complete answer regions and ordered step-level subregions, with question- and step-level boxes under a hierarchical constraint.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.25491","pdf":"https://arxiv.org/pdf/2606.25491","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.25491"},"evidence":{"snippet":"We introduce HG-Bench, a benchmark of 500 human-annotated K-12 homework samples curated from a 1,489,278-image source pool, with question-level and step-level boxes linked by a hierarchical containment constraint.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.25491"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HG-Bench evaluates page-aware, two-level answer-region grounding: given multi-page handwritten homework images, models must localize complete answer regions and ordered step-level subregions, with question- and step-level boxes under a hierarchical constraint.","whyItMatters":"Automated homework assessment needs both answer recognition and spatial grounding of reasoning steps. Prior benchmarks miss page-aware, multi-level localization; HG-Bench provides a reproducible protocol to measure this capability gap across models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"39c9fa1786310ab51bdd5cc4043c56cafc1d7cb28dc7a2124c791f16256ecc89"},"motivation":"Automated homework assessment depends not only on recognizing student answers, but also on accurately locating where each answer and each intermediate reasoning step appears in noisy, multi-page handwritten work.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.25491","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_e065ced955caaca5","familyId":"catalog_family_e065ced955caaca5","name":"HiddenMath","oneLine":"Google DeepMind's internal mathematical reasoning benchmark that introduces novel problems not encountered during model training to evaluate true mathematical reasoning capabilities rather than memorization","description":"Google DeepMind's internal mathematical reasoning benchmark that introduces novel problems not encountered during model training to evaluate true mathematical reasoning capabilities rather than memorization","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/hiddenmath","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e065ced955caaca5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/hiddenmath"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"hiddenmath","url":"https://llm-stats.com/benchmarks/hiddenmath","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":13,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_hightide_5b4de3e6","familyId":"bmf_909da4bf70ec","name":"HighTide","oneLine":"HighTide is an AI-assisted VLSI benchmark suite with open-source designs, Bazel-based compilation, and agent skills for design curation.","area":"Language & Knowledge","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Semiconductors"],"capabilities":[],"topics":["Agents"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04126","pdf":"https://arxiv.org/pdf/2606.04126","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04126"},"evidence":{"snippet":"We introduce HighTide, an evolving AI-assisted benchmark suite.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04126"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HighTide is an AI-assisted VLSI benchmark suite with open-source designs, Bazel-based compilation, and agent skills for design curation.","whyItMatters":"It may provide a reproducible testbed for evaluating AI in chip design, but its evaluation contract and reuse path are not yet clear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e90919a01fce5b6043c31d6d9d9e0c0a068ff72b6851d1014f3052c2d8669cd8"},"motivation":"We introduce HighTide, an evolving AI-assisted benchmark suite.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04126","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_beam-wise-statistical-background-subtracti_69023b31","familyId":"bmf_f58064b0727b","name":"HighwayScene","oneLine":"HighwayScene is a multi-LiDAR dataset recorded in a static roadside setup for evaluating background subtraction, with public annotations and implementations released.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.14868","pdf":"https://arxiv.org/pdf/2608.14868","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"This paper presents a comparative benchmark of beam-wise statistical background subtraction for statically mounted LiDAR sensors.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","benchmark study on existing or public datasets without a named artifact release"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14868"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"HighwayScene is a multi-LiDAR dataset recorded in a static roadside setup for evaluating background subtraction, with public annotations and implementations released.","whyItMatters":"It provides reproducible evaluation for static roadside LiDAR background subtraction, addressing a gap in cross-sensor comparisons.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"ebcc7b5f1a312349a25760a240bbc2e0014ed734b151929e90ee21c73a864c9b"},"motivation":"Background subtraction is a key preprocessing step for infrastructure-based LiDAR perception, enabling efficient isolation of dynamic traffic participants without semantic annotations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The abstract names the dataset and states that datasets, annotations, and implementations are publicly released, supporting reuse.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce HighwayScene, a new multi-LiDAR dataset recorded in a static roadside setup"},"publication":{"status":"acceptance_claimed","venue":"publication at the 2026 IEEE 29th International Conference on Intelligent Transportation Systems (ITSC), Naples, Italy","evidence":"Accepted for publication at the 2026 IEEE 29th International Conference on Intelligent Transportation Systems (ITSC), Naples, Italy, September 15-18, 2026","evidenceUrl":"https://arxiv.org/abs/2608.14868","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-19T11:05:13.395391Z"},"venueAttempts":[{"venueName":"publication at the 2026 IEEE 29th International Conference on Intelligent Transportation Systems (ITSC), Naples, Italy","reviewStatus":"accepted","decisionRaw":"Accepted for publication at the 2026 IEEE 29th International Conference on Intelligent Transportation Systems (ITSC), Naples, Italy, September 15-18, 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.14868","observedAt":"2026-08-19T11:05:13.395391Z","rawValue":"Accepted for publication at the 2026 IEEE 29th International Conference on Intelligent Transportation Systems (ITSC), Naples, Italy, September 15-18, 2026","level":"author-claim"}]}],"attentionForecast":{"score":35,"confidence":"Low","horizon":"7d","reason":"The dataset targets a specialized ITS application and lacks a public project URL in the provided evidence, limiting forecast confidence."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_himed_f6226988","familyId":"bmf_3f1df78e7570","name":"HiMed","oneLine":"HiMed is a Hindi medical dataset and benchmark suite covering Western and Indian medicine. It evaluates medical reasoning in Hindi across multiple tasks, with data and evaluation scripts publicly released on GitHub.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24635","pdf":"https://arxiv.org/pdf/2605.24635","project":null,"code":"https://github.com/FreedomIntelligence/HiMed","data":null,"hfPaper":"https://huggingface.co/papers/2605.24635"},"evidence":{"snippet":"To this end, we introduce HiMed, a Hindi reasoning medical corpus and benchmark suite covering both Western and Indian medicine.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24635"},"ranking":{},"description":"HiMed is a Hindi medical dataset and benchmark suite covering Western and Indian medicine. It evaluates medical reasoning in Hindi across multiple tasks, with data and evaluation scripts publicly released on GitHub.","whyItMatters":"Hindi remains underrepresented in medical LLMs; this benchmark enables systematic evaluation of cross-lingual medical reasoning, helping reduce performance gaps between English and Hindi in healthcare applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a880022ffd5b04467b15cdcddb91166da04621c5be64ce934bd883033151aacc"},"motivation":"Medical large language models hold promise for reducing healthcare disparities, yet Hindi remains severely underrepresented.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24635","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FreedomIntelligence","organizationType":"academic-lab","sourceUrl":"https://github.com/FreedomIntelligence/HiMed","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_c2c4f815df1512b3","familyId":"catalog_family_c2c4f815df1512b3","name":"HiPhO","oneLine":"HiPhO is a high-school physics olympiad benchmark evaluating multimodal reasoning over physics problems that include diagrams and figures.","description":"HiPhO is a high-school physics olympiad benchmark evaluating multimodal reasoning over physics problems that include diagrams and figures.","area":"Multimodal","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Science","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/hipho","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c2c4f815df1512b3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/hipho"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"hipho","url":"https://llm-stats.com/benchmarks/hipho","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","science","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_ba4c41e5ab9ad2d6","familyId":"catalog_family_ba4c41e5ab9ad2d6","name":"HLE w/ tools","oneLine":"Tool-augmented Humanity's Last Exam scores reported in DeepSeek-V4 thinking-mode evaluations.","description":"Tool-augmented Humanity's Last Exam scores reported in DeepSeek-V4 thinking-mode evaluations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ba4c41e5ab9ad2d6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/hlewithtools"}],"catalogSources":[{"catalog":"benchlm","sourceId":"hleWithTools","url":"https://benchlm.ai/benchmarks/hlewithtools","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"Humanity's Last Exam with tools","format":"Pass@1","tasks":"Expert questions with tool use","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_3248f91add0ab00d","familyId":"catalog_family_3248f91add0ab00d","name":"HLE w/o tools","oneLine":"Tool-free variant of Humanity's Last Exam that isolates a model's raw frontier reasoning.","description":"Tool-free variant of Humanity's Last Exam that isolates a model's raw frontier reasoning.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3248f91add0ab00d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/hlenotools"}],"catalogSources":[{"catalog":"benchlm","sourceId":"hleNoTools","url":"https://benchlm.ai/benchmarks/hlenotools","paperUrl":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","year":"2026","fullName":"Humanity's Last Exam without tools","format":"Tool-free expert QA","tasks":"Expert-level questions","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e90605fef1281a39","familyId":"catalog_family_e90605fef1281a39","name":"HLE-Verified","oneLine":"HLE-Verified evaluates multidisciplinary expert reasoning on a verified subset of Humanity's Last Exam.","description":"HLE-Verified evaluates multidisciplinary expert reasoning on a verified subset of Humanity's Last Exam.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/hle-verified","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e90605fef1281a39"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/hle-verified"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"hle-verified","url":"https://llm-stats.com/benchmarks/hle-verified","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_hll_52c829e2","familyId":"bmf_f97ab7d41b89","name":"HLL","oneLine":"HLL is a benchmark that evaluates multimodal agents on interactive CAPTCHA verification in a closed-loop GUI environment, covering diverse task types such as text transcription, image selection, sliders, jigsaw puzzles, and logic-based challenges. Scoring is based on task completion and trace-conditioned validation.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02449","pdf":"https://arxiv.org/pdf/2606.02449","project":null,"code":"https://github.com/XinhaoS0101/HLL","data":null,"hfPaper":"https://huggingface.co/papers/2606.02449"},"evidence":{"snippet":"We introduce \\textbf{Humanity's Last Line of Verification (HLL)}, a controlled benchmark that uses interactive CAPTCHA verification to evaluate whether agents can cross this boundary through grounded, human-like interaction rather than recognition alone.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02449"},"ranking":{},"description":"HLL is a benchmark that evaluates multimodal agents on interactive CAPTCHA verification in a closed-loop GUI environment, covering diverse task types such as text transcription, image selection, sliders, jigsaw puzzles, and logic-based challenges. Scoring is based on task completion and trace-conditioned validation.","whyItMatters":"HLL addresses the gap in measuring agent capability at automation-protected workflows, providing a testbed for comparing progress in human-like interaction and process consistency.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3d31bcb2387de283f6c8469fb538a05a6a19fbf95c8790d3508dab3d7df214c1"},"motivation":"Multimodal agents are increasingly expected to operate interfaces on behalf of users, raising a central deployment question: can they truly substitute for humans in workflows that services deliberately protect against automation?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02449","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Xinhao S0101","organizationType":"community","sourceUrl":"https://github.com/XinhaoS0101/HLL","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_897457a59168a17d","familyId":"catalog_family_897457a59168a17d","name":"HMMT 2025","oneLine":"Harvard-MIT Mathematics Tournament 2025 - A prestigious student-organized mathematics competition for high school students featuring two tournaments (November 2025 at MIT and February 2026 at Harvard) with individual tests, team rounds, and guts rounds","description":"Harvard-MIT Mathematics Tournament 2025 - A prestigious student-organized mathematics competition for high school students featuring two tournaments (November 2025 at MIT and February 2026 at Harvard) with individual tests, team rounds, and guts rounds","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/hmmt-2025","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_897457a59168a17d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/hmmt-2025"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"hmmt-2025","url":"https://llm-stats.com/benchmarks/hmmt-2025","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math"],"catalogModelCount":33,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_f60444ce6d182818","familyId":"catalog_family_f60444ce6d182818","name":"HMMT Feb 2023","oneLine":"A prestigious high school mathematics competition hosted jointly by Harvard and MIT, featuring challenging problems across various mathematical disciplines.","description":"A prestigious high school mathematics competition hosted jointly by Harvard and MIT, featuring challenging problems across various mathematical disciplines.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.hmmt.org/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f60444ce6d182818"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/hmmt2023"}],"catalogSources":[{"catalog":"benchlm","sourceId":"hmmt2023","url":"https://benchlm.ai/benchmarks/hmmt2023","paperUrl":"https://www.hmmt.org/","year":"2023","fullName":"Harvard-MIT Mathematics Tournament February 2023","format":"Competition mathematics","tasks":"Tournament problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_6263d44f474b726a","familyId":"catalog_family_6263d44f474b726a","name":"HMMT Feb 2024","oneLine":"The 2024 February edition of the Harvard-MIT Mathematics Tournament, continuing the tradition of challenging high school mathematics competition.","description":"The 2024 February edition of the Harvard-MIT Mathematics Tournament, continuing the tradition of challenging high school mathematics competition.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.hmmt.org/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6263d44f474b726a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/hmmt2024"}],"catalogSources":[{"catalog":"benchlm","sourceId":"hmmt2024","url":"https://benchlm.ai/benchmarks/hmmt2024","paperUrl":"https://www.hmmt.org/","year":"2024","fullName":"Harvard-MIT Mathematics Tournament February 2024","format":"Competition mathematics","tasks":"Tournament problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_11f1b4cc43f9ae72","familyId":"catalog_family_11f1b4cc43f9ae72","name":"HMMT Feb 2025","oneLine":"The most recent February edition of the Harvard-MIT Mathematics Tournament, featuring the latest challenging problems in competitive mathematics.","description":"The most recent February edition of the Harvard-MIT Mathematics Tournament, featuring the latest challenging problems in competitive mathematics.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.hmmt.org/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_11f1b4cc43f9ae72"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/hmmt2025"},{"role":"benchlm","url":"https://benchlm.ai/benchmarks/hmmtfeb2025"}],"catalogSources":[{"catalog":"benchlm","sourceId":"hmmt2025","url":"https://benchlm.ai/benchmarks/hmmt2025","paperUrl":"https://www.hmmt.org/","year":"2025","fullName":"Harvard-MIT Mathematics Tournament February 2025","format":"Competition mathematics","tasks":"Tournament problems","successorKey":null},{"catalog":"benchlm","sourceId":"hmmtFeb2025","url":"https://benchlm.ai/benchmarks/hmmtfeb2025","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2025","fullName":"Harvard-MIT Mathematics Tournament February 2025","format":"Contest mathematics","tasks":"Competition math problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_12cab1af150533ab","familyId":"catalog_family_12cab1af150533ab","name":"HMMT Feb 2026","oneLine":"A February 2026 HMMT slice used in newer frontier-model math comparisons.","description":"A February 2026 HMMT slice used in newer frontier-model math comparisons.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_12cab1af150533ab"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/hmmtfeb2026"}],"catalogSources":[{"catalog":"benchlm","sourceId":"hmmtFeb2026","url":"https://benchlm.ai/benchmarks/hmmtfeb2026","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"Harvard-MIT Mathematics Tournament February 2026","format":"Contest mathematics","tasks":"Competition math problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_c16edb9a6159b2b6","familyId":"catalog_family_c16edb9a6159b2b6","name":"HMMT Feb 26","oneLine":"HMMT February 2026 is a math competition benchmark based on problems from the Harvard-MIT Mathematics Tournament, testing advanced mathematical problem-solving and reasoning.","description":"HMMT February 2026 is a math competition benchmark based on problems from the Harvard-MIT Mathematics Tournament, testing advanced mathematical problem-solving and reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/hmmt-feb-26","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c16edb9a6159b2b6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/hmmt-feb-26"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"hmmt-feb-26","url":"https://llm-stats.com/benchmarks/hmmt-feb-26","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":12,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_820f58e73430bcf2","familyId":"catalog_family_820f58e73430bcf2","name":"HMMT Nov 2025","oneLine":"A November 2025 HMMT slice for high-end mathematical reasoning comparisons.","description":"A November 2025 HMMT slice for high-end mathematical reasoning comparisons.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_820f58e73430bcf2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/hmmtnov2025"}],"catalogSources":[{"catalog":"benchlm","sourceId":"hmmtNov2025","url":"https://benchlm.ai/benchmarks/hmmtnov2025","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2025","fullName":"Harvard-MIT Mathematics Tournament November 2025","format":"Contest mathematics","tasks":"Competition math problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_123166a3ea230751","familyId":"catalog_family_123166a3ea230751","name":"HMMT25","oneLine":"Harvard-MIT Mathematics Tournament 2025 - A prestigious student-organized mathematics competition for high school students featuring two tournaments (November 2025 at MIT and February 2026 at Harvard) with individual tests, team rounds, and guts rounds","description":"Harvard-MIT Mathematics Tournament 2025 - A prestigious student-organized mathematics competition for high school students featuring two tournaments (November 2025 at MIT and February 2026 at Harvard) with individual tests, team rounds, and guts rounds","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/hmmt25","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_123166a3ea230751"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/hmmt25"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"hmmt25","url":"https://llm-stats.com/benchmarks/hmmt25","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math"],"catalogModelCount":25,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_hof-bench_be2a2445","familyId":"bmf_69c4fa77605b","name":"HoF-Bench","oneLine":"Evaluates vulnerability discovery in source code using 95 real CVEs across 8 repositories pinned at vulnerable commits, with a detector-blinded judge crediting findings that match code path, root cause, attack condition, and impact.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.27030","pdf":"https://arxiv.org/pdf/2607.27030","project":null,"code":"https://github.com/weareaisle/HoF-Bench","data":"https://huggingface.co/datasets/aisleinc/HoF-Bench","hfPaper":"https://huggingface.co/papers/2607.27030"},"evidence":{"snippet":"We introduce HoF-Bench (named after AISLE's public Hall of Fame), a benchmark built from 95 of these public AI-discovered CVEs across eight repositories pinned at vulnerable commits.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":258,"hfDatasetLikes":5},"source":{"type":"arxiv","id":"2607.27030"},"ranking":{"90d":{"score":36,"rank":188,"coverage":0.85,"confidence":"High","datasetDownloadRank":34,"datasetRankPopulation":66}},"description":"Evaluates vulnerability discovery in source code using 95 real CVEs across 8 repositories pinned at vulnerable commits, with a detector-blinded judge crediting findings that match code path, root cause, attack condition, and impact.","whyItMatters":"Provides a realistic test bed for comparing vulnerability scanners on rediscovery of known real-world vulnerabilities, with a strict scoring protocol and reusable dataset.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"806022f73a2c918564f8b111f9fdc1026a65bd72f6c8256ba6bd81f2b84b4854"},"motivation":"LLM-based analyzers have begun finding real vulnerabilities in mature open-source projects: AISLE's analyzer is credited with more than 280 CVEs across 78 projects, including OpenSSL, curl, and GnuTLS.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27030","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AISLE","organizationType":"company-research-lab","sourceUrl":"https://github.com/weareaisle/HoF-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_holocount_b6a34c73","familyId":"bmf_bd0b3a3f9aa2","name":"HoloCount","oneLine":"HoloCount evaluates multimodal large language models on visual counting tasks across three levels: semantic counting (atomic and property-based enumeration), analytical counting (logical composition via spatial and set-based reasoning), and robustness testing (adverse scenarios and grounded counter-priors). The benchmark uses a hierarchical taxonomy and provides a dataset for evaluation.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.06420","pdf":"https://arxiv.org/pdf/2607.06420","project":"https://mm-mvr.github.io/HoloCount/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.06420"},"evidence":{"snippet":"To address these limitations, we introduce HoloCount, a holistic and diagnostically rich benchmark structured around a three-level hierarchical taxonomy.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06420"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"HoloCount evaluates multimodal large language models on visual counting tasks across three levels: semantic counting (atomic and property-based enumeration), analytical counting (logical composition via spatial and set-based reasoning), and robustness testing (adverse scenarios and grounded counter-priors). The benchmark uses a hierarchical taxonomy and provides a dataset for evaluation.","whyItMatters":"Existing counting benchmarks fail to capture complex failure modes under logical constraints or adversarial conditions. HoloCount provides a structured diagnostic tool to assess MLLM quantitative precision, revealing performance gaps as tasks shift from perception to analytical reasoning, and guiding development of more grounded multimodal systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8a5aecf80c7de9ed617c8b644b55b7fd52e8c32f92a01db2e358afaad9661c72"},"motivation":"Visual counting is a fundamental pillar of multimodal intelligence, requiring a seamless integration of fine-grained grounding and spatial reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06420","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HoloCount Team","organizationType":"academic-lab","sourceUrl":"https://mm-mvr.github.io/HoloCount/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_hoosierhelp_ef4e5cac","familyId":"bmf_7c4d86f1a55e","name":"HoosierHelp","oneLine":"HoosierHelp is an interactive benchmark for evaluating LLM agents in social service navigation. Agents interact with simulated users, issue structured resource-search calls, and select final resources from 3,971 Indiana public social service resources. The evaluation focuses on constraint grounding and handling non-ideal user interactions.","area":"Robotics & Embodied AI","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.HC"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09946","pdf":"https://arxiv.org/pdf/2608.09946","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09946"},"evidence":{"snippet":"We introduce HoosierHelp, an interactive benchmark grounded in 3,971 Indiana public social service resources.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09946"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HoosierHelp is an interactive benchmark for evaluating LLM agents in social service navigation. Agents interact with simulated users, issue structured resource-search calls, and select final resources from 3,971 Indiana public social service resources. The evaluation focuses on constraint grounding and handling non-ideal user interactions.","whyItMatters":"Existing benchmarks do not capture the interaction complexity and constraint-grounding demands of social service navigation. HoosierHelp addresses this gap by simulating realistic user behaviors, providing a basis for assessing agent reliability in a high-stakes domain where incorrect resource recommendations can have serious consequences.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7dddef945121a3ef83d8243de18132073fdbbfc3feb16b02440cf1d0f465c4b6"},"motivation":"Social service navigation requires connecting help-seeking individuals to resources that satisfy their needs and specific constraints.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09946","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"general"},{"id":"bm_hoprefusalbench_06f4ed80","familyId":"bmf_4c9fe07200c3","name":"HopRefusalBench","oneLine":"HopRefusalBench evaluates refusal behavior of search-augmented language model agents on multi-hop questions that are unanswerable, covering three causes of unanswerability and three chain topologies.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.01358","pdf":"https://arxiv.org/pdf/2608.01358","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01358"},"evidence":{"snippet":"We introduce HopRefusalBench, the first controlled benchmark of refusal within multi-hop search, comprising 889 unanswerable questions constructed from KILT-grounded entity paths.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01358"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HopRefusalBench evaluates refusal behavior of search-augmented language model agents on multi-hop questions that are unanswerable, covering three causes of unanswerability and three chain topologies.","whyItMatters":"Abstention benchmarks typically focus on single-hop queries, leaving evaluation gaps for failures that emerge during multi-hop reasoning and retrieval. This benchmark targets that gap and provides metrics for diagnosing refusal performance, aiding in improving agent reliability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"12bd6a9f71fd24469eed0ad9abf572d8dc30ced861d286d537f1681e08a5de36"},"motivation":"Search-augmented large language model agents are increasingly capable of solving knowledge-intensive tasks, but their behavior when a multi-hop question is fundamentally unanswerable remains poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.01358","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d8b2de1a63b757a6","familyId":"catalog_family_d8b2de1a63b757a6","name":"HorizonMath","oneLine":"HorizonMath is an extremely difficult frontier mathematics benchmark designed to test the limits of mathematical reasoning on research-level and competition-beyond problems.","description":"HorizonMath is an extremely difficult frontier mathematics benchmark designed to test the limits of mathematical reasoning on research-level and competition-beyond problems.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/horizonmath","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d8b2de1a63b757a6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/horizonmath"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"horizonmath","url":"https://llm-stats.com/benchmarks/horizonmath","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_hounsbench_1139158f","familyId":"bmf_5e09e564fc52","name":"HounsBench","oneLine":"HounsBench is a CT-centric patient-state benchmark evaluating three task families: readout, reconstruction, and simulation. It provides patient-disjoint splits and per-family metrics for evaluating models on volumetric medical images and clinical language.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12904","pdf":"https://arxiv.org/pdf/2608.12904","project":null,"code":"https://github.com/byhwhite/HounsWorld.git","data":null,"hfPaper":"https://huggingface.co/papers/2608.12904"},"evidence":{"snippet":"To operationalize this view, we introduce HounsBench, a computed tomography (CT) centric patient-state benchmark that unifies these three task families with patient-disjoint splits and per-family metrics, and HounsWorld, a 3B multimodal world model that treats volumetric scans and language as observations of the shared state through Joint Understanding-Generation Learning.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12904"},"ranking":{"30d":{"score":34,"rank":66,"coverage":0.55,"confidence":"Low"},"90d":{"score":33,"rank":214,"coverage":0.55,"confidence":"Low"}},"description":"HounsBench is a CT-centric patient-state benchmark evaluating three task families: readout, reconstruction, and simulation. It provides patient-disjoint splits and per-family metrics for evaluating models on volumetric medical images and clinical language.","whyItMatters":"HounsBench addresses the evaluation gap for CT-centered intelligence by unifying diverse tasks under a shared patient-state inference framework, enabling assessment of models that must integrate imaging and language for clinical decision support.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4b246dd78ed2bf04c1d377f54fe27c005aff81346d8028d094ad13239f99da87"},"motivation":"Clinical intelligence requires estimating a patient's underlying condition from incomplete observations rather than learning isolated mappings from scans to answers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12904","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HounsWorld Project","organizationType":"academic-lab","sourceUrl":"https://github.com/byhwhite/HounsWorld.git","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_6b53a3168cb7f356","familyId":"catalog_family_6b53a3168cb7f356","name":"HR-Bench (4k)","oneLine":"HR-Bench (4k) evaluates image understanding on high-resolution visual inputs with a 4k setting.","description":"HR-Bench (4k) evaluates image understanding on high-resolution visual inputs with a 4k setting.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/hr-bench-4k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6b53a3168cb7f356"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/hr-bench-4k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"hr-bench-4k","url":"https://llm-stats.com/benchmarks/hr-bench-4k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_hribench_c65825aa","familyId":"bmf_9b6d4c3d73e5","name":"HRIBench","oneLine":"HRIBench evaluates intent-aware human-robot collaboration through structured scenario scripts covering roles of Instructor, Collaborator, and Intruder, with 13 tasks and 650 episodes, scoring synchronization, responsiveness, protocol compliance, and safety.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.13056","pdf":"https://arxiv.org/pdf/2607.13056","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.13056"},"evidence":{"snippet":"To address this gap, we introduce HRIBench, a diagnostic benchmark for intent-aware human-robot collaboration based on executable interaction scenarios.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.13056"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"HRIBench evaluates intent-aware human-robot collaboration through structured scenario scripts covering roles of Instructor, Collaborator, and Intruder, with 13 tasks and 650 episodes, scoring synchronization, responsiveness, protocol compliance, and safety.","whyItMatters":"Existing VLA benchmarks focus on isolated manipulation, leaving a gap in evaluating coordination under shared agency. HRIBench provides a standardized protocol for assessing temporal coordination and intent understanding, with evidence that fine-tuning on its data improves real-world task success.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7e1e7e0b42abdc53a3334f4312f2089ded4d9900e0e8dc77aa433188ecd3dd9d"},"motivation":"Current vision-language-action (VLA) benchmarks primarily evaluate isolated manipulation skills while leaving human-robot interaction structure largely unmodeled.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.13056","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HRIBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.13056","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_dd199ce379f8fb3a","familyId":"catalog_family_dd199ce379f8fb3a","name":"HRM8K","oneLine":"Korean mathematical reasoning (high-school to Olympiad level).","description":"Korean mathematical reasoning (high-school to Olympiad level).","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Korean"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://benchlm.ai/benchmarks/hrm8k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dd199ce379f8fb3a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/hrm8k"}],"catalogSources":[{"catalog":"benchlm","sourceId":"hrm8k","url":"https://benchlm.ai/benchmarks/hrm8k","paperUrl":null,"year":null,"fullName":"HAE-RAE Math 8K","format":"Math word problems","tasks":"8,011 instances","successorKey":null}],"catalogCategories":["korean"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ht-bench_4c10005b","familyId":"bmf_712886e53afa","name":"HT-Bench","oneLine":"HT-Bench is a multi-task benchmark for tactile representation learning with egocentric vision and full-hand tactile data, including retrieval, inpainting, synthesis, and prediction tasks.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19161","pdf":"https://arxiv.org/pdf/2606.19161","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.19161"},"evidence":{"snippet":"To this end, we introduce \\textbf{HT-Bench}, a large-scale multi-task benchmark for dexterous full-hand tactile sensing, comprising 10M RGB frames and 7.8M tactile frames collected across 226 tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19161"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"HT-Bench is a multi-task benchmark for tactile representation learning with egocentric vision and full-hand tactile data, including retrieval, inpainting, synthesis, and prediction tasks.","whyItMatters":"Tactile representation learning lacks universal benchmarks. HT-Bench explores egocentric vision paired with tactile data, providing a scalable evaluation direction for dexterous manipulation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8f8b040db355a7a4f2282ffeb970b6c4a5a58cc0dc8e02da61da86308b75c141"},"motivation":"Establishing a universal benchmark for tactile representation learning in robotic manipulation remains challenging due to the diversity of tactile sensor designs, data formats, and robot embodiments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19161","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_hug-vis_24e7fc8b","familyId":"bmf_3e006d284c40","name":"HUG-VIS","oneLine":"Visual intelligence seeks to perceive, interpret, and synthesize the visual world and is central to modern computer vision.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.26517","pdf":"https://arxiv.org/pdf/2608.26517","project":null,"code":"https://github.com/GML-MMGroup/HUG-VIS","data":null,"hfPaper":"https://huggingface.co/papers/2608.26517"},"evidence":{"snippet":"We address this gap with HUG-VIS, a unified benchmark for Human-centered Understanding and Generation in Visual Intelligence.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26517"},"ranking":{"30d":{"score":31,"rank":72,"coverage":0.55,"confidence":"Low"},"90d":{"score":31,"rank":227,"coverage":0.55,"confidence":"Low"}},"motivation":"Visual intelligence seeks to perceive, interpret, and synthesize the visual world and is central to modern computer vision.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.26517","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"lib_humaneval","familyId":"family_humaneval","name":"HumanEval","oneLine":"Established benchmark family · Code & Software.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code & Software"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2021-01-01","releaseDatePrecision":"year","firstRelease":{"year":2021,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2107.03374","pdf":null,"project":"https://github.com/openai/human-eval","code":"https://github.com/openai/human-eval","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_humaneval"},"ranking":{},"recordType":"family","aliases":[],"sourceAttribution":[{"role":"official-repository","url":"https://github.com/openai/human-eval"}],"adoptionRefs":["deepseek-v3"],"modelReportReferences":[{"sourceId":"deepseek-v3","url":"https://github.com/deepseek-ai/DeepSeek-V3","provider":"DeepSeek","model":"DeepSeek-V3"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"humaneval","url":"https://benchlm.ai/benchmarks/humaneval","paperUrl":"https://arxiv.org/abs/2107.03374","year":"2021","fullName":"Evaluating Large Language Models Trained on Code","format":"Python function generation","tasks":"164 problems","successorKey":null},{"catalog":"llm-stats","sourceId":"humaneval","url":"https://llm-stats.com/benchmarks/humaneval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false},{"catalog":"llm-stats","sourceId":"humaneval+","url":"https://llm-stats.com/benchmarks/humaneval+","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","reasoning","code"],"catalogModelCount":66,"catalogStarCount":0},{"id":"catalog_810807308d2673d1","familyId":"catalog_family_810807308d2673d1","name":"HumanEval Plus","oneLine":"Enhanced version of HumanEval that extends the original test cases by 80x using EvalPlus framework for rigorous evaluation of LLM-synthesized code functional correctness, detecting previously undetected wrong code","description":"Enhanced version of HumanEval that extends the original test cases by 80x using EvalPlus framework for rigorous evaluation of LLM-synthesized code functional correctness, detecting previously undetected wrong code","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/humaneval-plus","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_810807308d2673d1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/humaneval-plus"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"humaneval-plus","url":"https://llm-stats.com/benchmarks/humaneval-plus","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_12bd73d8ceec44dd","familyId":"catalog_family_12bd73d8ceec44dd","name":"HumanEval-Average","oneLine":"A variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics","description":"A variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/humaneval-average","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_12bd73d8ceec44dd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/humaneval-average"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"humaneval-average","url":"https://llm-stats.com/benchmarks/humaneval-average","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_db3cb8d38ce35983","familyId":"catalog_family_db3cb8d38ce35983","name":"HumanEval-ER","oneLine":"A variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics","description":"A variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/humaneval-er","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_db3cb8d38ce35983"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/humaneval-er"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"humaneval-er","url":"https://llm-stats.com/benchmarks/humaneval-er","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_1e92f1c27e2af05f","familyId":"catalog_family_1e92f1c27e2af05f","name":"HumanEval-Mul","oneLine":"A multilingual variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics","description":"A multilingual variant of the HumanEval benchmark that measures functional correctness for synthesizing programs from docstrings, consisting of 164 original programming problems assessing language comprehension, algorithms, and simple mathematics","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/humaneval-mul","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1e92f1c27e2af05f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/humaneval-mul"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"humaneval-mul","url":"https://llm-stats.com/benchmarks/humaneval-mul","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_bootstrapping-niche-multilingual-code-tran_de25b8dd","familyId":"bmf_30c196f82c84","name":"HumanEval-X++","oneLine":"HumanEval-X++ is an execution-based benchmark extending HumanEval-X to a broad many-to-many language space for code translation evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":0.45,"links":{"report":"https://arxiv.org/abs/2608.13854","pdf":"https://arxiv.org/pdf/2608.13854","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To evaluate the niche translation capability, we introduce HumanEval-X++, an execution-based benchmark that extends HumanEval-X to a broad many-to-many language space.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13854"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HumanEval-X++ is an execution-based benchmark extending HumanEval-X to a broad many-to-many language space for code translation evaluation.","whyItMatters":"It provides execution-validated evaluation for niche many-to-many code translation, where parallel supervision is sparse.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"b7deaa5a4d844955c4a953a082ef2c4b75d009a9f52dfdfbee6b2d3c5abd40fc"},"motivation":"Code translation must preserve executable behavior across many programming languages, yet neural code translation has largely focused on a few popular languages such as C++, Java, and Python.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The abstract formally names the benchmark and describes an execution-based scoring contract.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce HumanEval-X++, an execution-based benchmark that extends HumanEval-X to a broad many-to-many language space"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13854","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":40,"confidence":"Low","horizon":"7d","reason":"The benchmark extends an existing known suite but lacks explicit public release details, making early attention uncertain."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_6da692bf12f42b89","familyId":"catalog_family_6da692bf12f42b89","name":"HumanEvalFIM-Average","oneLine":"Average evaluation of HumanEval Fill-in-the-Middle benchmark variants (single-line, multi-line, random-span) for assessing code infilling capabilities of language models","description":"Average evaluation of HumanEval Fill-in-the-Middle benchmark variants (single-line, multi-line, random-span) for assessing code infilling capabilities of language models","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/humanevalfim-average","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6da692bf12f42b89"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/humanevalfim-average"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"humanevalfim-average","url":"https://llm-stats.com/benchmarks/humanevalfim-average","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"lib_humanitys_last_exam","familyId":"family_humanitys_last_exam","name":"Humanity's Last Exam","oneLine":"Established benchmark family · Knowledge & Reasoning.","area":"Knowledge & Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge & Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2025-01-01","releaseDatePrecision":"year","firstRelease":{"year":2025,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2501.14249","pdf":null,"project":"https://lastexam.ai/","code":"https://github.com/centerforaisafety/hle","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_humanitys_last_exam"},"ranking":{},"recordType":"family","aliases":["HLE"],"sourceAttribution":[{"role":"official-project","url":"https://lastexam.ai/"}],"adoptionRefs":["openai-gpt5","anthropic-claude4","google-gemini25"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"humanity's-last-exam","url":"https://llm-stats.com/benchmarks/humanity's-last-exam","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning","vision"],"catalogModelCount":101,"catalogStarCount":0},{"id":"catalog_c258daebcc7b06f4","familyId":"catalog_family_c258daebcc7b06f4","name":"Humanity's Last Exam (no tools, text-only)","oneLine":"Text-only Humanity's Last Exam variant evaluated without tool use.","description":"Text-only Humanity's Last Exam variant evaluated without tool use.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/humanity's-last-exam-(no-tools,-text-only)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c258daebcc7b06f4"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/humanity's-last-exam-(no-tools,-text-only)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"humanity's-last-exam-(no-tools,-text-only)","url":"https://llm-stats.com/benchmarks/humanity's-last-exam-(no-tools,-text-only)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_bb0cc3176b036740","familyId":"catalog_family_bb0cc3176b036740","name":"Humanity's Last Exam (with tools, text-only)","oneLine":"Text-only Humanity's Last Exam variant evaluated with tool use enabled.","description":"Text-only Humanity's Last Exam variant evaluated with tool use enabled.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/humanity's-last-exam-(with-tools,-text-only)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bb0cc3176b036740"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/humanity's-last-exam-(with-tools,-text-only)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"humanity's-last-exam-(with-tools,-text-only)","url":"https://llm-stats.com/benchmarks/humanity's-last-exam-(with-tools,-text-only)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_humanmovevqa_d7611820","familyId":"bmf_21e6a36c45a9","name":"HumanMoveVQA","oneLine":"HumanMoveVQA evaluates video MLLMs on reasoning about human trajectory and orientation changes in videos, using a first-frame anchored world coordinate system and 10K question-answer pairs across seven reasoning categories.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.27999","pdf":"https://arxiv.org/pdf/2606.27999","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.27999"},"evidence":{"snippet":"We introduce HumanMoveVQA, the first comprehensive benchmark designed to evaluate global trajectory and orientation reasoning from an exocentric perspective.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.27999"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HumanMoveVQA evaluates video MLLMs on reasoning about human trajectory and orientation changes in videos, using a first-frame anchored world coordinate system and 10K question-answer pairs across seven reasoning categories.","whyItMatters":"Existing benchmarks fail to probe global human motion in space over time; HumanMoveVQA targets this gap, but without accessible data or code its practical value for model comparison remains unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5bf1227adaba81623a969cd0b3dc8ad8fdb45c08f7403f652ddcd2b4b724b9fa"},"motivation":"Despite the rapid advance of Multimodal Large Language Models (MLLMs) in high-level video understanding, a fundamental bottleneck remains: these models collapse complex human motion into coarse semantic labels.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.27999","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_humanoidarena_033240fc","familyId":"bmf_0b58750af400","name":"HumanoidArena","oneLine":"HumanoidArena is a simulation benchmark for egocentric hierarchical whole-body learning, evaluating high-level policies that predict whole-body actions for low-level general motion trackers across seven leg-critical human-object and human-scene interaction tasks, with perturbation-conditioned and GMT-conditioned evaluation.","area":"Robotics & Embodied AI","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.17833","pdf":"https://arxiv.org/pdf/2606.17833","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.17833"},"evidence":{"snippet":"We introduce HumanoidArena, a simulation-first benchmark for egocentric hierarchical whole-body learning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":18,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.17833"},"ranking":{"90d":{"score":50,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"HumanoidArena is a simulation benchmark for egocentric hierarchical whole-body learning, evaluating high-level policies that predict whole-body actions for low-level general motion trackers across seven leg-critical human-object and human-scene interaction tasks, with perturbation-conditioned and GMT-conditioned evaluation.","whyItMatters":"Existing benchmarks rarely evaluate the policy-tracker interface itself, leaving open whether intermediate whole-body actions are executable, robust under task distribution shifts, and transferable across different GMT backends. HumanoidArena addresses this gap by emphasizing leg-critical interactions and transferable intermediate action representations, providing a basis for comparing hierarchical control architectures.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"95278c9da6f9ab9bedbb13e4d0f0fd5034f758a3253809662f4a4e1d9f1b0cea"},"motivation":"Humanoid robots promise whole-body interaction in human-centered environments, but scalable policy learning remains difficult because task-level decision-making and whole-body dynamic execution are tightly coupled.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.17833","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"general"},{"id":"bm_humanoidvln_60fdcf09","familyId":"bmf_622a7fe4192c","name":"HumanoidVLN","oneLine":"Evaluates vision-language navigation for humanoid robots across four embodiments in physics-grounded simulator scenarios. Includes 933 episodes with instructions and multiple stylistic variants, assessing success rate and normalized Dynamic Time Warping.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Multimodal"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12860","pdf":"https://arxiv.org/pdf/2608.12860","project":"https://humanoid-vln.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12860"},"evidence":{"snippet":"We present HumanoidVLN, a physics-grounded simulator and benchmark for VLN across diverse humanoid embodiments.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12860"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates vision-language navigation for humanoid robots across four embodiments in physics-grounded simulator scenarios. Includes 933 episodes with instructions and multiple stylistic variants, assessing success rate and normalized Dynamic Time Warping.","whyItMatters":"Addresses the gap in VLN benchmarks for bipedal locomotion and diverse morphologies, providing a platform to compare navigation models under physical constraints and supporting sim-to-real transfer studies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"25e0707a2701b98970ab65cf7e8042cee518170d985ba6ae841a31c825289576"},"motivation":"Vision-Language Navigation (VLN) for humanoid robots poses challenges existing benchmarks fail to address: bipedal locomotion imposes physical constraints absent from wheeled agents, humanoid morphologies vary across platforms, and egocentric observations are distorted by locomotion-induced camera dynamics.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12860","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HumanoidVLN team","organizationType":"benchmark-organization","sourceUrl":"https://humanoid-vln.github.io/","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_hush-bench_669096c5","familyId":"bmf_9824d9a002a6","name":"HUSH-Bench","oneLine":"HUSH-Bench evaluates conversational agents' use of sensitive history under a conservative policy, with 2,400 prompts and matched no-memory references, measuring unsolicited integration and memory access.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.06055","pdf":"https://arxiv.org/pdf/2606.06055","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.06055"},"evidence":{"snippet":"We introduce HUSH-Bench, a controlled benchmark of 2,400 benign prompts paired with histories containing one marked sensitive disclosure and matched no-memory references.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06055"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HUSH-Bench evaluates conversational agents' use of sensitive history under a conservative policy, with 2,400 prompts and matched no-memory references, measuring unsolicited integration and memory access.","whyItMatters":"It isolates memory retention from use in dialogue generation, showing that retrieval systems may surface sensitive data even without user request. This motivates separating storage, retrieval, and scope decisions in memory design.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6c9b8b24af5d4c4d5ed74589f0eff744c9f7dcdeb6b5afb9d2eb55c49067053c"},"motivation":"Long-term memory helps conversational agents maintain continuity across sessions, while relevance and current-turn warrant remain distinct decisions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.06055","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_hy-multiturn_edce774f","familyId":"bmf_394a3bb3e96c","name":"Hy-MultiTurn","oneLine":"Hy-MultiTurn is a Chinese benchmark for deep multi-turn dialogue understanding, with six controlled evaluation modes covering constraint memory, object localization, and action suppression across 12-76 turn dialogues.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.29196","pdf":"https://arxiv.org/pdf/2607.29196","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.29196"},"evidence":{"snippet":"To address these limitations, we analyze real chatbot failures to identify six recurring mechanisms and use them to define six controlled evaluation modes in Hy-MultiTurn, a Chinese benchmark for deep multi-turn dialogue understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.29196"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Hy-MultiTurn is a Chinese benchmark for deep multi-turn dialogue understanding, with six controlled evaluation modes covering constraint memory, object localization, and action suppression across 12-76 turn dialogues.","whyItMatters":"Evaluates long multi-turn dialogue capabilities that existing benchmarks miss, offering controlled modes for diagnosing model failures.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c1fbf03fe6a286a621e4a0fd9065deff22a6f54896d0d6c9569b0f9d733c6444"},"motivation":"Long-running multi-turn interactions with chatbots and agents are now common, and a correct response often depends on remembering earlier details, tracking later revisions, identifying intended objects or referents, and withholding action when required conditions are unmet.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.29196","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_hybridcodeauthorship_7e92f56f","familyId":"bmf_bcaf12963fb7","name":"HybridCodeAuthorship","oneLine":"HybridCodeAuthorship is a benchmark of Python files with interleaved human- and AI-authored lines, built from CodeSearchNet, for developing and evaluating AI-generated code detection algorithms at line- and chunk-level.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.12620","pdf":"https://arxiv.org/pdf/2606.12620","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.12620"},"evidence":{"snippet":"To fill these gaps, we introduce HybridCodeAuthorship, a novel benchmark of Python code files with interleaved human- and AI-authored lines of code to simulate authentic utilization of AI code assistants.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.12620"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HybridCodeAuthorship is a benchmark of Python files with interleaved human- and AI-authored lines, built from CodeSearchNet, for developing and evaluating AI-generated code detection algorithms at line- and chunk-level.","whyItMatters":"Addresses the need for realistic benchmarks for detecting AI-generated code in industry codebases, which is important for risk management and productivity analysis. Provides a challenging testbed for detection algorithms.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fc855bb09779b5652619bf4db361935122f3f53543be6fad8ea2cacec60c9b9c"},"motivation":"Thanks to the rapid adoption of AI code assistants powered by large language models (LLMs), industry codebases are, increasingly, a hybrid of AI- and human-authored code.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"publication_reported","venue":"LREC 2026 proceedings (pp. 1520-1532)","evidence":"LREC 2026 proceedings (pp. 1520-1532)","evidenceUrl":"https://arxiv.org/abs/2606.12620","source":"arxiv-journal-reference","evidenceLevel":"strong-author-metadata","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publications":[{"venueName":"LREC 2026 proceedings (pp. 1520-1532)","publicationStatus":"published","evidence":[{"sourceType":"arxiv-journal-reference","sourceUrl":"https://arxiv.org/abs/2606.12620","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"LREC 2026 proceedings (pp. 1520-1532)","level":"strong-author-metadata"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_hyperalign-bench_25471ecc","familyId":"bmf_6933ddf622b3","name":"HyperAlign-Bench","oneLine":"HyperAlign-Bench is a benchmark introduced in a paper for evaluating hypergraph structural modeling in large language models, covering vertex-level and hyperedge-level tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.21858","pdf":"https://arxiv.org/pdf/2605.21858","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.21858"},"evidence":{"snippet":"To systematically evaluate different methods in hypergraph structural modeling, we introduce HyperAlign-Bench.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.21858"},"ranking":{},"description":"HyperAlign-Bench is a benchmark introduced in a paper for evaluating hypergraph structural modeling in large language models, covering vertex-level and hyperedge-level tasks.","whyItMatters":"It addresses the gap in evaluating LLMs on high-order relational structures, which are common in real-world data but underrepresented in existing benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bbee5089f77780139c899f01fbd5876b685c31f302c6389345630109906ca587"},"motivation":"Large language models (LLMs) have recently shown strong potential in modeling relational structures.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.21858","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_hyperimagenet_06535482","familyId":"bmf_aa3a815479d6","name":"HyperImageNet","oneLine":"HyperImageNet is a dataset of 26,084 airborne hyperspectral image patches with 224 spectral bands and 138 fine-grained land-cover categories, providing raw imagery, pixel-level semantic labels, and object-level instance masks for semantic and instance segmentation, along with an open-environment evaluation protocol.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.21050","pdf":"https://arxiv.org/pdf/2607.21050","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.21050"},"evidence":{"snippet":"We present HyperImageNet, a large-scale benchmark for fine-grained hyperspectral land-cover understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.21050"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"HyperImageNet is a dataset of 26,084 airborne hyperspectral image patches with 224 spectral bands and 138 fine-grained land-cover categories, providing raw imagery, pixel-level semantic labels, and object-level instance masks for semantic and instance segmentation, along with an open-environment evaluation protocol.","whyItMatters":"Existing hyperspectral benchmarks lack fine-grained categories and instance-level annotations; HyperImageNet enables evaluation of models on high-spatial-resolution imagery with strict spatial separation for open-environment generalization.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2f52e8aa95da4627200049213411900aa877d5beb7e35d47f46ebb75516cea7f"},"motivation":"We present HyperImageNet, a large-scale benchmark for fine-grained hyperspectral land-cover understanding.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.21050","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HyperImageNet Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.21050","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_hypershadow_31f52aa8","familyId":"bmf_21616075ae21","name":"HyperShadow","oneLine":"HyperShadow evaluates binary classification of 3D point clouds as either native 3D objects or 3D projections of objects in 4-6 spatial dimensions, with static and temporal tracks, four corruption tiers, and fixed train/eval splits.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14419","pdf":"https://arxiv.org/pdf/2607.14419","project":null,"code":"https://github.com/AkshaySasi/hypershadow","data":"https://huggingface.co/datasets/AkshaySasi/hypershadow","hfPaper":"https://huggingface.co/papers/2607.14419"},"evidence":{"snippet":"We introduce HyperShadow, the first public benchmark in which the fourth, fifth, and sixth dimensions are spatial: the task is to decide whether a 3D point cloud is a native three-dimensional shape or the projection, the \"shadow\", of a rigid object living in R^N (N = 4-6).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":209,"hfDatasetLikes":1},"source":{"type":"arxiv","id":"2607.14419"},"ranking":{"90d":{"score":18,"rank":392,"coverage":1.0,"confidence":"High","datasetDownloadRank":36,"datasetRankPopulation":66}},"description":"HyperShadow evaluates binary classification of 3D point clouds as either native 3D objects or 3D projections of objects in 4-6 spatial dimensions, with static and temporal tracks, four corruption tiers, and fixed train/eval splits.","whyItMatters":"Addresses a gap in benchmarks for high-dimensional geometric data, providing a controlled testbed for studying projection signatures and out-of-distribution detection without physical reality claims.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0bc2b545df5e1b8bf15654455da6198b2e780d8904503c0da26df7ef632cdb80"},"motivation":"Machine-learning datasets labelled \"4D\" universally denote three spatial dimensions plus time.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14419","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AkshaySasi","organizationType":"community","sourceUrl":"https://github.com/AkshaySasi/hypershadow","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_3fcadca8bbe60d1d","familyId":"catalog_family_3fcadca8bbe60d1d","name":"Hypersim","oneLine":"Hypersim evaluates 3D grounding and depth understanding in synthetic indoor scenes.","description":"Hypersim evaluates 3D grounding and depth understanding in synthetic indoor scenes.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Spatial Reasoning","3D","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/hypersim","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3fcadca8bbe60d1d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/hypersim"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"hypersim","url":"https://llm-stats.com/benchmarks/hypersim","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["spatial reasoning","3d","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_i-webgenbench_56eafa1f","familyId":"bmf_6c242fe4bdad","name":"I-WebGenBench","oneLine":"A benchmark of 19 research papers with expert-built interactive systems for evaluating agents that convert PDFs into executable web applications.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00750","pdf":"https://arxiv.org/pdf/2606.00750","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.00750"},"evidence":{"snippet":"To evaluate this task, we introduce a benchmark of 19 research papers paired with expert-built interactive systems as ground truth.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00750"},"ranking":{},"description":"A benchmark of 19 research papers with expert-built interactive systems for evaluating agents that convert PDFs into executable web applications.","whyItMatters":"Supports evaluation of interactive system generation but relies on limited data and lacks documented public reuse.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6ba319af5acbf6434db3bcbd5c01a60e96db0c91ec97cea9d99e7a89ec54e587"},"motivation":"Recent advances in visual language models have enabled autonomous agents for complex reasoning, tool use, and document understanding.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00750","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_iba-bench_e7767345","familyId":"bmf_cb530590be3a","name":"IBA-Bench","oneLine":"IBA-Bench evaluates LLM agents on implicit behavioral alignment using longitudinal interaction histories, across nine application domains.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02171","pdf":"https://arxiv.org/pdf/2608.02171","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.02171"},"evidence":{"snippet":"To address this challenge, we introduce IBA-Bench, a benchmark for implicit behavioral alignment constructed from longitudinal interaction histories that contain noise, implicit cues, and temporal inconsistencies.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02171"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"IBA-Bench evaluates LLM agents on implicit behavioral alignment using longitudinal interaction histories, across nine application domains.","whyItMatters":"Assesses whether agents can infer and satisfy implicit user constraints, a practical gap in personalization.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0a819d20cf1eb2a54f8a1067fdb7cfbc53be89103e7063a85298bb8749cca3d6"},"motivation":"Large Language Models have enabled increasingly capable autonomous agents, yet personalization remains critical for making such agents practically useful.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02171","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_icae-bench_617decda","familyId":"bmf_0115589cfafb","name":"ICAE-Bench","oneLine":"ICAE-Bench evaluates coding agents on interactive project-building tasks. Agents receive a fuzzy product requirement and must clarify missing details via an automated user agent, then implement the project in a container. Scoring uses black-box tests and multi-dimensional diagnostics including functional correctness, semantic/API similarity, structural fidelity, design quality, and interaction quality.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.21217","pdf":"https://arxiv.org/pdf/2607.21217","project":null,"code":"https://github.com/ALEX-nlp/ICAE-EVAL","data":null,"hfPaper":"https://huggingface.co/papers/2607.21217"},"evidence":{"snippet":"In this paper, we introduce ICAE-Bench, a benchmark for evaluating coding agents under interactive project-building settings.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.21217"},"ranking":{"90d":{"score":35,"rank":192,"coverage":0.7,"confidence":"Medium"}},"description":"ICAE-Bench evaluates coding agents on interactive project-building tasks. Agents receive a fuzzy product requirement and must clarify missing details via an automated user agent, then implement the project in a container. Scoring uses black-box tests and multi-dimensional diagnostics including functional correctness, semantic/API similarity, structural fidelity, design quality, and interaction quality.","whyItMatters":"Existing coding benchmarks focus on static, fully specified tasks, leaving a gap for interactive, open-ended development. ICAE-Bench provides a reproducible protocol for measuring agent performance in transforming incomplete requirements into working software, which is increasingly relevant for real-world coding workflows.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1e44db164316f5d9ccafc513c5c1c94a95245d9fe5bf9c7fe2046f647e0c03fa"},"motivation":"The recent emergence of vibe-coding workflows is changing what coding agents are expected to do.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.21217","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ALEX-nlp","organizationType":"academic-lab","sourceUrl":"https://github.com/ALEX-nlp/ICAE-EVAL","role":"benchmark-publisher"}],"capabilityGroups":["Agents","Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_15ccd58064dbed4b","familyId":"catalog_family_15ccd58064dbed4b","name":"IDE-Bench","oneLine":"An 80-task software-engineering benchmark across eight repositories that tests whether autonomous IDE agents can explore, edit, run, and verify code changes end to end.","description":"An 80-task software-engineering benchmark across eight repositories that tests whether autonomous IDE agents can explore, edit, run, and verify code changes end to end.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2601.20886","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_15ccd58064dbed4b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/idebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"ideBench","url":"https://benchlm.ai/benchmarks/idebench","paperUrl":"https://arxiv.org/abs/2601.20886","year":"2026","fullName":"IDE-Bench","format":"Autonomous IDE-agent task completion (pass@1)","tasks":"80 tasks across 8 repositories","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_ideagene-bench_3d711bf3","familyId":"bmf_04604847c861","name":"IdeaGene-Bench","oneLine":"Evaluates scientific lineage reasoning and lineage-grounded idea generation through two tracks: closed-form IG-Exam and generation IG-Arena with Population-Evolution Score.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.08758","pdf":"https://arxiv.org/pdf/2607.08758","project":null,"code":"https://github.com/VisionXLab/IdeasHaveGenomes","data":null,"hfPaper":"https://huggingface.co/papers/2607.08758"},"evidence":{"snippet":"We present IdeaGene-Bench (IG-Bench), a benchmark for scientific lineage reasoning and lineage-grounded idea generation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":41,"hfDailySubmittedAt":"2026-07-10T00:00:00.000Z","githubStars":31,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.08758"},"ranking":{"90d":{"score":51,"rank":59,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates scientific lineage reasoning and lineage-grounded idea generation through two tracks: closed-form IG-Exam and generation IG-Arena with Population-Evolution Score.","whyItMatters":"Fills the gap in evaluating AI systems' understanding of scientific idea evolution, with results showing a compositional bottleneck in current models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"65157e110ddb54f4fe771e238bef226360b968653ef0c8afe82164cac8faaef1"},"motivation":"Scientific ideas rarely start from a blank page.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.08758","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"VisionXLab","organizationType":"academic-lab","sourceUrl":"https://github.com/VisionXLab/IdeasHaveGenomes","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ideal-bench_dab673c0","familyId":"bmf_b9aa0b64dd23","name":"IDEAL-Bench","oneLine":"IDEAL-Bench evaluates Vision-Language Models on holistic 3D layout inference from single images of indoor scenes, scoring predictions across five numerical dimensions and a perceptual render-and-compare protocol. It uses a procedurally generated dataset of 1,000 re-renderable Blender environments.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Geometric reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.03614","pdf":"https://arxiv.org/pdf/2607.03614","project":"https://ideal3d.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.03614"},"evidence":{"snippet":"To this end, we introduce IDEAL-Bench, an evaluation suite that requires VLMs to predict structured 3D layouts on photorealistic indoor scenes across 10 room types, scored along five numerical dimensions and a perceptual render-and-compare protocol.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.03614"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"IDEAL-Bench evaluates Vision-Language Models on holistic 3D layout inference from single images of indoor scenes, scoring predictions across five numerical dimensions and a perceptual render-and-compare protocol. It uses a procedurally generated dataset of 1,000 re-renderable Blender environments.","whyItMatters":"Current VLM spatial evaluation relies on question answering, which misses structural understanding. IDEAL-Bench provides a reproducible, quantitative assessment of geometric and structural competencies, revealing model weaknesses in measuring scenes rather than describing them, and offering a diagnostic for genuine spatial intelligence.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3c2efb906e1db2c52ce9292af6a1fcfee7c4a6e02a7b8289ced2d1b156bb41f0"},"motivation":"Spatial question answering is the dominant paradigm for evaluating spatial intelligence in Vision-Language Models (VLMs), but it leaves a complementary axis of spatial competence under-evaluated: holistic 3D layout inference, which predicts every visible object's pose and extent from a single image in a structured form.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.03614","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"IDEAL-Bench Team","organizationType":"academic-lab","sourceUrl":"https://ideal3d.github.io/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_idiolink_97ac8121","familyId":"bmf_f52bc82b7c8c","name":"IdioLink","oneLine":"IdioLink is a retrieval benchmark with 10,700 documents and 2,140 queries across 107 idioms, testing the linking of idiomatic expressions to literal or paraphrased equivalents. Annotations mark core meaning spans, and scoring is based on retrieval quality against these spans.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22247","pdf":"https://arxiv.org/pdf/2605.22247","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.22247"},"evidence":{"snippet":"We introduce IdioLink, a retrieval benchmark designed to test whether models can link idiomatic expressions to conceptually equivalent meanings expressed in literal or paraphrased forms.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.22247"},"ranking":{},"description":"IdioLink is a retrieval benchmark with 10,700 documents and 2,140 queries across 107 idioms, testing the linking of idiomatic expressions to literal or paraphrased equivalents. Annotations mark core meaning spans, and scoring is based on retrieval quality against these spans.","whyItMatters":"Existing retrieval models often rely on surface similarity and fail on idiomatic expressions. IdioLink provides a reusable testbed to measure semantic abstraction beyond lexical overlap, offering a practical way to evaluate idiom-aware retrieval systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cd2be64596d629add0bf0d855d62f18c00dac323bf38b70e7ffb9d4351dfdbf0"},"motivation":"Idioms pose a fundamental challenge for language models, as their meaning cannot be inferred from surface form alone.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.22247","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_idp-bench_f6df4252","familyId":"bmf_c03dcd02a45c","name":"IDP-Bench","oneLine":"IDP-Bench evaluates large language models on interdependent privacy scenarios, covering recognition of co-ownership, identification of contextual integrity parameters, and judgments of sharing appropriateness.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-06","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.09908","pdf":"https://arxiv.org/pdf/2606.09908","project":null,"code":"https://github.com/tisl-lab/Interdependent_Privacy_Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.09908"},"evidence":{"snippet":"We address this gap by introducing \\textbf{IDP-Bench}: the first LLM benchmark for IDP scenarios, grounded in the Contextual Integrity (CI) framework.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09908"},"ranking":{"90d":{"score":28,"rank":283,"coverage":0.55,"confidence":"Low"}},"description":"IDP-Bench evaluates large language models on interdependent privacy scenarios, covering recognition of co-ownership, identification of contextual integrity parameters, and judgments of sharing appropriateness.","whyItMatters":"Interdependent privacy is a critical yet underexplored risk when LLMs act as personal assistants; a reusable benchmark enables systematic comparison and improvement of models' privacy reasoning in shared-data contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6de12f0be02d1f103a75c882aa71c5e42385c223a4214625b8c5f6d8eb1f0d8c"},"motivation":"Large language models (LLMs) are becoming widely deployed as personal AI assistants with access to sensitive user data, making privacy a major challenge for their design and evaluation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09908","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TISL Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/tisl-lab/Interdependent_Privacy_Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_935f68319d4f227e","familyId":"catalog_family_935f68319d4f227e","name":"IF","oneLine":"Instruction-Following Evaluation (IFEval) benchmark for large language models, focusing on verifiable instructions with 25 types of instructions and around 500 prompts containing one or more verifiable constraints","description":"Instruction-Following Evaluation (IFEval) benchmark for large language models, focusing on verifiable instructions with 25 types of instructions and around 500 prompts containing one or more verifiable constraints","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Structured Output","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/if","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_935f68319d4f227e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/if"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"if","url":"https://llm-stats.com/benchmarks/if","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["structured output","general"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general"},{"id":"catalog_ff770459cce56302","familyId":"catalog_family_ff770459cce56302","name":"IFBench","oneLine":"Instruction Following Benchmark evaluating model's ability to follow complex instructions","description":"Instruction Following Benchmark evaluating model's ability to follow complex instructions","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Instructionfollowing","Instruction Following","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://benchlm.ai/benchmarks/ifbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ff770459cce56302"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/ifbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ifbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"ifBench","url":"https://benchlm.ai/benchmarks/ifbench","paperUrl":null,"year":2025,"fullName":"Instruction Following Benchmark","format":null,"tasks":58,"successorKey":null},{"catalog":"llm-stats","sourceId":"ifbench","url":"https://llm-stats.com/benchmarks/ifbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["instructionFollowing","instruction following","general"],"catalogModelCount":35,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general"},{"id":"bm_ifcmemorybench_f2c4e9c3","familyId":"bmf_c1b4184d38fa","name":"IFCMemoryBench","oneLine":"IFCMemoryBench evaluates long-term memory in LLM-based agents for BIM information retrieval, with 143 multi-session tasks across 19 projects and 4,016 prior sessions, requiring integration of remembered context with live IFC queries.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.IR"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.26072","pdf":"https://arxiv.org/pdf/2607.26072","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.26072"},"evidence":{"snippet":"We introduce IFCMemoryBench, a benchmark for evaluating long-term memory in LLM-based BIM information retrieval.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26072"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"IFCMemoryBench evaluates long-term memory in LLM-based agents for BIM information retrieval, with 143 multi-session tasks across 19 projects and 4,016 prior sessions, requiring integration of remembered context with live IFC queries.","whyItMatters":"Existing memory evaluations focus on conversational recall, not professional domains. IFCMemoryBench provides a test for whether agents can reuse information across sessions in a structured, domain-specific environment, revealing gaps in current memory systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"130a6741595f56737952fc8d1620137b90d8b59ba67b1586753db1347b86e2a4"},"motivation":"Long-term memory is becoming a core capability of LLM-based agents, but existing evaluations largely test conversational recall in open-domain or persona-grounded settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26072","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"lib_ifeval","familyId":"family_ifeval","name":"IFEval","oneLine":"Established benchmark family · Instruction Following.","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Instruction Following"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2023-01-01","releaseDatePrecision":"year","firstRelease":{"year":2023,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2311.07911","pdf":null,"project":"https://github.com/google-research/google-research/tree/master/instruction_following_eval","code":"https://github.com/google-research/google-research/tree/master/instruction_following_eval","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_ifeval"},"ranking":{},"recordType":"family","aliases":["Instruction-Following Eval"],"sourceAttribution":[{"role":"official-repository","url":"https://github.com/google-research/google-research/tree/master/instruction_following_eval"}],"adoptionRefs":["deepseek-v3","google-gemini25"],"modelReportReferences":[{"sourceId":"deepseek-v3","url":"https://github.com/deepseek-ai/DeepSeek-V3","provider":"DeepSeek","model":"DeepSeek-V3"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"ifeval","url":"https://benchlm.ai/benchmarks/ifeval","paperUrl":"https://arxiv.org/abs/2311.07911","year":"2023","fullName":"Instruction-Following Eval","format":"Constrained generation","tasks":"541 prompts across 25 instruction types","successorKey":null},{"catalog":"llm-stats","sourceId":"ifeval","url":"https://llm-stats.com/benchmarks/ifeval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["instructionFollowing","instruction following","structured output","general"],"catalogModelCount":68,"catalogStarCount":0},{"id":"bm_ifhierbench_c5c33d5e","familyId":"bmf_ed9dc64cf457","name":"IFHierBench","oneLine":"IFHierBench is a hierarchical instruction-following benchmark with 600 prompts and deterministic checkers, evaluating LLMs on satisfying constraints at different output scopes. It measures prompt-level accuracy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27912","pdf":"https://arxiv.org/pdf/2607.27912","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.27912"},"evidence":{"snippet":"We introduce IFHierBench, a hierarchical instruction-following benchmark of 600 prompts stratified across four constraint-tree depths and 35 distinct constraints, each prompt paired with a deterministic checker that verifies satisfaction at every scope.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27912"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"IFHierBench is a hierarchical instruction-following benchmark with 600 prompts and deterministic checkers, evaluating LLMs on satisfying constraints at different output scopes. It measures prompt-level accuracy.","whyItMatters":"Instruction-following is critical for LLM deployment, and existing benchmarks treat constraints flatly. This benchmark addresses a gap by evaluating nested constraints, but its availability is unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"970ea146b7affeba1c5fb611bbbad1463fc3ad2523d37a82ce16266a53fa58c7"},"motivation":"Instruction-following ability is critical for deploying large language models in real-world applications, where downstream components depend on the output satisfying specific constraints.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27912","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ifmtbench_b83f8c4f","familyId":"bmf_c2644d034b9f","name":"IFMTBench","oneLine":"Evaluates multilingual translation systems on instruction following across seven languages, six constraint types, and compositional multi-constraint requests.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2605.28218","pdf":"https://arxiv.org/pdf/2605.28218","project":null,"code":"https://github.com/Tencent-Hunyuan/Hy-MT2/tree/main/IFMTBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.28218"},"evidence":{"snippet":"We introduce \\bench, a benchmark for multilingual translation instruction following covering seven languages, with 4,506 single-constraint and 2,838 multi-constraint items spanning six constraint dimensions and five compositional patterns with instructions issued in all seven languages.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":571,"githubScope":"hosting_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28218"},"ranking":{},"description":"Evaluates multilingual translation systems on instruction following across seven languages, six constraint types, and compositional multi-constraint requests.","whyItMatters":"Shows whether translation systems can follow terminology, style, length, audience, and other workflow constraints while preserving translation quality.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8ffde504a4f6f0125b70268f3de10201211ac5b384be2b71235cdc817cda7118"},"motivation":"Modern translation workflows demand more than semantic equivalence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28218","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Tencent Hunyuan","organizationType":"company-research-lab","role":"benchmark-publisher","sourceUrl":"https://github.com/Tencent-Hunyuan/Hy-MT2/tree/main/IFMTBench"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ih-benchmark_f309c884","familyId":"bmf_505ebf5f9c8e","name":"IH-Benchmark","oneLine":"IH-Benchmark evaluates instruction-hierarchy robustness in LLMs via conflicting instructions from system, user, and tool outputs, covering 44 constraint families across five domains with a binary pass/fail protocol.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Robustness"],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.25987","pdf":"https://arxiv.org/pdf/2607.25987","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.25987"},"evidence":{"snippet":"We present IH-Benchmark, a conflict-centered benchmark for instruction-hierarchy robustness across direct system-user conflicts (S>U) and tool-mediated user-tool (U>T) conflicts.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25987"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"IH-Benchmark evaluates instruction-hierarchy robustness in LLMs via conflicting instructions from system, user, and tool outputs, covering 44 constraint families across five domains with a binary pass/fail protocol.","whyItMatters":"Existing instruction-hierarchy benchmarks cover limited conflict types and tool interactions. IH-Benchmark provides a systematic evaluation across conflict surfaces, constraint types, and attack presentations, revealing that robustness is not a single capability but a set of behaviors with distinct failure modes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1ed86f357979990217f303f2c9d031227465185497921e25e38937b00e5da8f6"},"motivation":"When a language model receives conflicting instructions from different priority levels, which one does it actually follow?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25987","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_ihbench_0d580e81","familyId":"bmf_a132a2f51250","name":"IHBench","oneLine":"IHBench evaluates post-interruption recovery in voice agents executing state-machine-driven workflows across 10 enterprise domains. It scores task fulfillment and recovery quality for six interruption types.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19595","pdf":"https://arxiv.org/pdf/2606.19595","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.19595"},"evidence":{"snippet":"We introduce IHBench (Interruption Handling Benchmark), a benchmark that evaluates post-interruption recovery in voice agents executing state-machine-driven workflows across 10 enterprise domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19595"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"IHBench evaluates post-interruption recovery in voice agents executing state-machine-driven workflows across 10 enterprise domains. It scores task fulfillment and recovery quality for six interruption types.","whyItMatters":"Voice agents must handle interruptions while maintaining workflow progress, but existing benchmarks measure only interruption timing. IHBench focuses on recovery quality, a distinct capability axis important for deployed agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"820eb257129b74e3b1d30c3a5e6d9c9f47961bfb3aba352fcbe4ac6b79eedc01"},"motivation":"Voice agents deployed in structured workflows (customer service, healthcare scheduling, account management) must handle frequent user interruptions while maintaining progress through multi-step procedures.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19595","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_ii-bench_6645329e","familyId":"bmf_e098118d24e6","name":"II-Bench","oneLine":"II-Bench evaluates computer-use agents against low-harm adversarial tasks across three platforms, with 444 examples and a testing framework.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02018","pdf":"https://arxiv.org/pdf/2608.02018","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.02018"},"evidence":{"snippet":"To systematically investigate this blind spot, we present II-Bench, a collection of seemingly harmless adversarial tasks.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02018"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"II-Bench evaluates computer-use agents against low-harm adversarial tasks across three platforms, with 444 examples and a testing framework.","whyItMatters":"Exposes security blind spots in human-in-the-loop defenses for computer-use agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7a0b6cccfe47f257d527480101c0136c923e632f86faa905e8e1e140fa873ce0"},"motivation":"Computer-use agents (CUAs), which empower large language models to autonomously operate operating systems and the web, are increasingly vulnerable to indirect prompt injection attacks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02018","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_92baf138f8914ddd","familyId":"catalog_family_92baf138f8914ddd","name":"Image2FloorPlan","oneLine":"Image2FloorPlan is an in-house benchmark evaluating multimodal models on generating structured floor plans and interactive frontends directly from images such as design mockups and room photos.","description":"Image2FloorPlan is an in-house benchmark evaluating multimodal models on generating structured floor plans and interactive frontends directly from images such as design mockups and room photos.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Code","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/image2floorplan","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_92baf138f8914ddd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/image2floorplan"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"image2floorplan","url":"https://llm-stats.com/benchmarks/image2floorplan","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","code","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_087ec402f23016a9","familyId":"catalog_family_087ec402f23016a9","name":"ImageMining","oneLine":"ImageMining evaluates multimodal models on extracting structured information from images using tool use, measuring ability to combine visual understanding with tool-based retrieval and analysis.","description":"ImageMining evaluates multimodal models on extracting structured information from images using tool use, measuring ability to combine visual understanding with tool-based retrieval and analysis.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Agents","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_087ec402f23016a9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/imagemining"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/imagemining"}],"catalogSources":[{"catalog":"benchlm","sourceId":"imageMining","url":"https://benchlm.ai/benchmarks/imagemining","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"ImageMining","format":"Image-grounded retrieval and extraction","tasks":"Visual retrieval tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"imagemining","url":"https://llm-stats.com/benchmarks/imagemining","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","agents","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_imaging-101_1fefb66e","familyId":"bmf_1df9e5402412","name":"Imaging-101","oneLine":"Imaging-101 evaluates coding agents on 57 computational imaging tasks across six scientific domains, with three tracks for planning, function-level unit tests, and end-to-end reconstruction.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.10789","pdf":"https://arxiv.org/pdf/2607.10789","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.10789"},"evidence":{"snippet":"We introduce Imaging-101, a benchmark of 57 expert-verified computational imaging tasks spanning six scientific domains, each grounded in a peer-reviewed paper and canonicalized into a standardized four-stage pipeline (preprocessing, forward physics modeling, inverse solver, and visualization) Three evaluation tracks (planning, function-level unit tests, and end-to-end reconstruction) probe distinct agent capabilities across the full pipeline.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.10789"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Imaging-101 evaluates coding agents on 57 computational imaging tasks across six scientific domains, with three tracks for planning, function-level unit tests, and end-to-end reconstruction.","whyItMatters":"General coding benchmarks may not capture domain-specific challenges in scientific imaging, and this benchmark could help assess agent capabilities in algorithm selection, physical conventions, and pipeline integration.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c883469cf9d0a86629947ee0c300addad297d829ad6d3654eb9cffd8598205d6"},"motivation":"Computational imaging, which recovers hidden signals from indirect, noisy measurements, underpins quantitative discovery across scientific disciplines, yet building a correct reconstruction pipeline demands deep domain expertise and remains laborious even for domain scientists.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.10789","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_imagingbench_e9f34716","familyId":"bmf_960f5481dc88","name":"ImagingBench","oneLine":"ImagingBench evaluates agentic AI systems on 20 computational imaging tasks spanning ray and wave optics, image signal processing, inverse reconstruction, computational sensing, and calibration, across three settings: Expert, Planner, and Forward.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.07189","pdf":"https://arxiv.org/pdf/2607.07189","project":"https://cirp-lab.github.io/imagingbench","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.07189"},"evidence":{"snippet":"We present ImagingBench, a benchmark of 20 computational imaging tasks spanning five categories: ray and wave optics, image signal processing, inverse reconstruction, computational sensing, and calibration.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.07189"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ImagingBench evaluates agentic AI systems on 20 computational imaging tasks spanning ray and wave optics, image signal processing, inverse reconstruction, computational sensing, and calibration, across three settings: Expert, Planner, and Forward.","whyItMatters":"Reveals the gap between semantic visual competence and physically grounded imaging performance, providing a unified testbed to measure progress in agentic AI for computational imaging.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4f4a63ed0f2afe3e4c0f49b95c07e0baf417a36737e3d86c1dfb8388b35a0dcd"},"motivation":"Vision-language models (VLMs) and agentic AI have shown strong performance on semantic visual tasks, but it remains unclear whether they can handle the physics and inverse problems that underlie computational imaging.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.07189","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_imbench_d224375e","familyId":"bmf_7d389515cfd2","name":"IMBench","oneLine":"IMBench evaluates intuitive robotic manipulation, combining perception, physical reasoning, action generation, and execution across 35 tasks and 14K trajectories.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning","Robot manipulation"],"topics":["Robotics","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.15641","pdf":"https://arxiv.org/pdf/2607.15641","project":"https://imbench.org/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.15641"},"evidence":{"snippet":"We introduce IMBENCH, a benchmark designed to evaluate intuitive manipulation as an integrated capability spanning perception, physical reasoning, action generation, and iterative execution.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.15641"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"IMBench evaluates intuitive robotic manipulation, combining perception, physical reasoning, action generation, and execution across 35 tasks and 14K trajectories.","whyItMatters":"Could provide a more holistic benchmark for robotic manipulation, but details on public access and evaluation protocol are unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d82ba7069530c355226e5fb73b20157e03a8315593502be67ffa3ab4fc0bc8a6"},"motivation":"Humans combine reasoning and motor control to solve complex manipulation tasks under diverse constraints.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"SemRob Workshop, RSS 2026","evidence":"Accepted to SemRob Workshop, RSS 2026. Project Website: https://imbench.org/","evidenceUrl":"https://arxiv.org/abs/2607.15641","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"SemRob Workshop, RSS 2026","reviewStatus":"accepted","decisionRaw":"Accepted to SemRob Workshop, RSS 2026. Project Website: https://imbench.org/","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.15641","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to SemRob Workshop, RSS 2026. Project Website: https://imbench.org/","level":"author-claim"}]}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_imcbench_04071d33","familyId":"bmf_420ea5d66c7d","name":"IMCBench","oneLine":"IMCBench evaluates multimodal LLMs in image-grounded, multi-turn medical conversations, scoring safety, accuracy, and uncertainty use on a 1-5 scale.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.28556","pdf":"https://arxiv.org/pdf/2606.28556","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.28556"},"evidence":{"snippet":"To address this gap, we introduce IMCBench, an image-grounded, multi-turn medical conversation benchmark that pairs real, publicly available clinical images with synthetic patient profiles to simulate realistic patient-clinician interactions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.28556"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"IMCBench evaluates multimodal LLMs in image-grounded, multi-turn medical conversations, scoring safety, accuracy, and uncertainty use on a 1-5 scale.","whyItMatters":"Addresses the gap in medical AI benchmarks by combining clinical images with multi-turn dialogue, enabling assessment of diagnostic accuracy alongside patient safety and uncertainty management.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2fef12f4c3f66365e3506c1fbdcd60c65c5e5fa36c7811ad5d18a4b1aa727fda"},"motivation":"Recent advances in large language models and vision-language models have enabled reasoning over multimodal data, offering opportunities for clinical applications such as decision support and triaging.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECML PKDD 2026","evidence":"Accepted at ECML PKDD 2026. 22 pages, 2 figures","evidenceUrl":"https://arxiv.org/abs/2606.28556","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECML PKDD 2026","reviewStatus":"accepted","decisionRaw":"Accepted at ECML PKDD 2026. 22 pages, 2 figures","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.28556","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at ECML PKDD 2026. 22 pages, 2 figures","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_7ceb189d6fece95f","familyId":"catalog_family_7ceb189d6fece95f","name":"IMO 2025","oneLine":"IMO 2025 evaluates models on the six problems from the 2025 International Mathematical Olympiad, requiring rigorous proof-based reasoning. Following the MathArena methodology, proofs are graded against human expert rubrics by dual strong judge models, with the minimum of the two scores taken as final. Maximum score is 42 points.","description":"IMO 2025 evaluates models on the six problems from the 2025 International Mathematical Olympiad, requiring rigorous proof-based reasoning. Following the MathArena methodology, proofs are graded against human expert rubrics by dual strong judge models, with the minimum of the two scores taken as final. Maximum score is 42 points.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/imo-2025","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7ceb189d6fece95f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/imo-2025"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"imo-2025","url":"https://llm-stats.com/benchmarks/imo-2025","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_7cdf21e11f345f5a","familyId":"catalog_family_7cdf21e11f345f5a","name":"IMO 2026","oneLine":"Proof-based olympiad performance on all six IMO 2026 problems.","description":"Proof-based olympiad performance on all six IMO 2026 problems.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7cdf21e11f345f5a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/imo2026"}],"catalogSources":[{"catalog":"benchlm","sourceId":"imo2026","url":"https://benchlm.ai/benchmarks/imo2026","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"International Mathematical Olympiad 2026","format":"Official-style proof score","tasks":"6 proof-based problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_9f633d8c67b1c89b","familyId":"catalog_family_9f633d8c67b1c89b","name":"IMOAnswerBench","oneLine":"IMO-AnswerBench is a benchmark for evaluating mathematical reasoning capabilities on International Mathematical Olympiad (IMO) problems, focusing on answer generation and verification.","description":"IMO-AnswerBench is a benchmark for evaluating mathematical reasoning capabilities on International Mathematical Olympiad (IMO) problems, focusing on answer generation and verification.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9f633d8c67b1c89b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/imoanswerbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/imo-answerbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"imoAnswerBench","url":"https://benchlm.ai/benchmarks/imoanswerbench","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"IMOAnswerBench","format":"Pass@1 math benchmark","tasks":"Advanced mathematical answer generation","successorKey":null},{"catalog":"llm-stats","sourceId":"imo-answerbench","url":"https://llm-stats.com/benchmarks/imo-answerbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":20,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_50e3578d8fbe5eba","familyId":"catalog_family_50e3578d8fbe5eba","name":"IMOProof-Adv","oneLine":"IMOProof-Adv is an advanced benchmark of International Mathematical Olympiad-style proof problems requiring rigorous multi-step mathematical proofs.","description":"IMOProof-Adv is an advanced benchmark of International Mathematical Olympiad-style proof problems requiring rigorous multi-step mathematical proofs.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/imoproof-adv","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_50e3578d8fbe5eba"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/imoproof-adv"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"imoproof-adv","url":"https://llm-stats.com/benchmarks/imoproof-adv","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_imug-bench_282ebfce","familyId":"bmf_92e790a6df66","name":"IMUG-Bench","oneLine":"IMUG-Bench evaluates unified multimodal models on multi-turn interleaved image-text understanding and generation. It includes 3,113 samples and 12,034 interaction turns across Static Spatial, Temporal Causal, and Hybrid classes, with dynamic understanding questions.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09169","pdf":"https://arxiv.org/pdf/2606.09169","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.09169"},"evidence":{"snippet":"To bridge this gap, we propose IMUG-Bench, a comprehensive benchmark for multi-turn interleaved image-text dialogue of UMMs that jointly evaluates their understanding and generation capabilities.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09169"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"IMUG-Bench evaluates unified multimodal models on multi-turn interleaved image-text understanding and generation. It includes 3,113 samples and 12,034 interaction turns across Static Spatial, Temporal Causal, and Hybrid classes, with dynamic understanding questions.","whyItMatters":"Existing benchmarks fail to evaluate multi-turn interleaved interactions and expose bias. IMUG-Bench provides a comprehensive evaluation revealing capability boundaries and failure modes, and explores test-time scaling strategies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ce091803bc0a1f379bbd008d46778b66cba1c3da185d636365d11f760279076c"},"motivation":"In recent years, unified multimodal models (UMMs) have emerged to support both understanding and generation within a single framework.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09169","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_c0cbff04cc90f5b4","familyId":"catalog_family_c0cbff04cc90f5b4","name":"INCLUDE","oneLine":"Include benchmark - specific documentation not found in official sources","description":"Include benchmark - specific documentation not found in official sources","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multilingual","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c0cbff04cc90f5b4"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/include"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/include"}],"catalogSources":[{"catalog":"benchlm","sourceId":"include","url":"https://benchlm.ai/benchmarks/include","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"INCLUDE","format":"Multilingual benchmark","tasks":"Cross-lingual understanding","successorKey":null},{"catalog":"llm-stats","sourceId":"include","url":"https://llm-stats.com/benchmarks/include","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multilingual","general"],"catalogModelCount":31,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_include-bench_e92f5105","familyId":"bmf_8801a3a5bd87","name":"INCLUDE-BENCH","oneLine":"Evaluates disability-related bias in text-to-image models using 119K generated images across bias dimensions and contexts, with the Stereotype Content Model Score.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.08515","pdf":"https://arxiv.org/pdf/2607.08515","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.08515"},"evidence":{"snippet":"To address this, we introduce INCLUDE-BENCH, the first large-scale benchmark for evaluating disability-related bias in T2I models.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.08515"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates disability-related bias in text-to-image models using 119K generated images across bias dimensions and contexts, with the Stereotype Content Model Score.","whyItMatters":"Addresses the underexplored area of disability stereotypes in T2I models, providing a large-scale evaluation for representational harms.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"37e065f941ad815e9e2913dfd28709c6916cf52b0f3f3d23d44c087eed4208a1"},"motivation":"Text-to-image (T2I) models have been shown to exhibit social biases.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.08515","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_21964c78509e994b","familyId":"catalog_family_21964c78509e994b","name":"independence-bench","oneLine":"Catalog-listed benchmark; original-source verification is pending.","description":"Catalog-listed benchmark; original-source verification is pending.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/community:07c9946d-dcf0-4977-a640-a6b1356b4f0b","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_21964c78509e994b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/community:07c9946d-dcf0-4977-a640-a6b1356b4f0b"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"community:07c9946d-dcf0-4977-a640-a6b1356b4f0b","url":"https://llm-stats.com/benchmarks/community:07c9946d-dcf0-4977-a640-a6b1356b4f0b","datasetSlug":"independence-bench","versionCount":2,"subsetCount":1,"rowCount":null,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":["general"],"catalogModelCount":5,"catalogStarCount":1,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_indic-diarbench_c5a01338","familyId":"bmf_ad88e51e7898","name":"Indic DiarBench","oneLine":"Multilingual joint diarization and ASR benchmark for 22 Indian languages, with ~108 hours of human-corrected multi-speaker audio from meetings, far-field, and in-the-wild sources, including code-mixing and overlaps.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.23808","pdf":"https://arxiv.org/pdf/2607.23808","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.23808"},"evidence":{"snippet":"In this work, we introduce Indic DiarBench, a speaker diarization and ASR benchmark dataset spanning all 22 scheduled languages of India.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23808"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Multilingual joint diarization and ASR benchmark for 22 Indian languages, with ~108 hours of human-corrected multi-speaker audio from meetings, far-field, and in-the-wild sources, including code-mixing and overlaps.","whyItMatters":"Provides a standardized evaluation suite for speaker diarization and ASR on Indian languages, addressing a gap in multilingual speech technology assessment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"059b892921ca680858a0c22ceb66a8b681f85e60fb8bf68b2ce88a3f198b9401"},"motivation":"In this work, we introduce Indic DiarBench, a speaker diarization and ASR benchmark dataset spanning all 22 scheduled languages of India.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.23808","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Indic DiarBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.23808","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_industrial-dexterity-benchmark_210a10b0","familyId":"bmf_35062a925b27","name":"Industrial Dexterity Benchmark","oneLine":"The Industrial Dexterity Benchmark (IDB) provides physical boards and tasks for industrial dexterous manipulation, including cable management, cable harness, and gearbox assembly. It evaluates a robot's ability to perform tasks such as cable insertion and grasping via end-to-end imitation learning policies.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14021","pdf":"https://arxiv.org/pdf/2607.14021","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.14021"},"evidence":{"snippet":"As a part of this work, we introduce three key contributions: a set of Industrial Dexterity Benchmark (IDB) boards aimed to mimic datacenter cable management, automotive cable harnesses, and gearbox assembly tasks; a scalable imitation learning framework (DAG-ROS); and a multimodal diffusion-based policy framework (AG-iDP3) that creates models fusing RGB images, point clouds, joint positions, and wrist-frame wrench data.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14021"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"The Industrial Dexterity Benchmark (IDB) provides physical boards and tasks for industrial dexterous manipulation, including cable management, cable harness, and gearbox assembly. It evaluates a robot's ability to perform tasks such as cable insertion and grasping via end-to-end imitation learning policies.","whyItMatters":"The benchmark addresses the lack of standardized evaluation for industrial dexterous manipulation, offering a repeatable protocol with defined tasks and success metrics. It enables systematic comparison of learning-based versus classical control methods, aiding automation decisions for high up-time industrial environments.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2424e8b3066aae14fb49369473786dff7cbafad6b63ddc1bf338ec1d4ae6a5ca"},"motivation":"Dexterous manipulation remains a critical bottleneck in industrial automation; tasks such as cable routing, connector insertion, and precision assembly still rely heavily on manual labor despite decades of robotics research.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14021","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Industrial Dexterity Benchmark Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.14021","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_industrybench-mipu_6704f5ae","familyId":"bmf_6260f7570604","name":"IndustryBench-MIPU","oneLine":"IndustryBench-MIPU evaluates MLLMs on extracting structured attribute-value pairs from multi-image industrial product data, covering text recognition, visual reasoning, and cross-image integration.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.14383","pdf":"https://arxiv.org/pdf/2606.14383","project":null,"code":"https://github.com/alibaba-multimodal-industrial-ai/IndustryBench-MIPU","data":null,"hfPaper":"https://huggingface.co/papers/2606.14383"},"evidence":{"snippet":"To fill this gap, we introduce IndustryBench-MIPU, the first large-scale benchmark for multi-image industrial product understanding, built around structured attribute extraction -- recovering property-value pairs from product images.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":4,"hfDailySubmittedAt":"2026-06-18T00:00:00.000Z","githubStars":12,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.14383"},"ranking":{"90d":{"score":39,"rank":163,"coverage":0.7,"confidence":"Medium"}},"description":"IndustryBench-MIPU evaluates MLLMs on extracting structured attribute-value pairs from multi-image industrial product data, covering text recognition, visual reasoning, and cross-image integration.","whyItMatters":"Industrial product specifications are scattered across images, and current MLLMs show a recall gap; this benchmark quantifies multi-image understanding limits for procurement and safety.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f62a1ce1637e2cb638f2f123a1f25a2591353c98cd163fea25419178a97024a7"},"motivation":"Industrial products such as valves and circuit breakers are defined by dense technical specifications that govern procurement, compatibility, and safety across supply chains.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.14383","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Alibaba","organizationType":"company-research-lab","sourceUrl":"https://github.com/alibaba-multimodal-industrial-ai/IndustryBench-MIPU","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_a85804a726de18ed","familyId":"catalog_family_a85804a726de18ed","name":"InferenceBench","oneLine":"A benchmark for open-ended LLM inference optimization by AI agents. Agents receive a base model, one H100, and a fixed time budget to build a valid OpenAI-compatible inference server that improves serving speed.","description":"A benchmark for open-ended LLM inference optimization by AI agents. Agents receive a base model, one H100, and a fixed time budget to build a valid OpenAI-compatible inference server that improves serving speed.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://inferencebench.ai/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a85804a726de18ed"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/inferencebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"inferenceBench","url":"https://benchlm.ai/benchmarks/inferencebench","paperUrl":"https://inferencebench.ai/","year":"2026","fullName":"InferenceBench","format":"Two-hour autonomous CLI agent run","tasks":"4 inference-serving optimization scenarios","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_inferq_453f0fc4","familyId":"bmf_429a10ce02b2","name":"InferQ","oneLine":"InferQ is a database-oriented benchmark for quantum circuit simulation, generating compositional circuits and emitting SQL workloads with feature extraction for workload characterization.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Quantum Technology"],"capabilities":[],"topics":["quant-ph"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.29134","pdf":"https://arxiv.org/pdf/2607.29134","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.29134"},"evidence":{"snippet":"We present InferQ, a database-oriented benchmark for quantum circuit simulation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.29134"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"InferQ is a database-oriented benchmark for quantum circuit simulation, generating compositional circuits and emitting SQL workloads with feature extraction for workload characterization.","whyItMatters":"Enables systematic database research on quantum simulation, supporting query optimization, physical design, and engine-level evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dd131eb241e99fb079763dc6679ad54b5390423fabcf40411c15a5b543e7aafc"},"motivation":"Recent work suggests that relational database management systems (RDBMSs) can execute quantum circuit simulation by compiling the simulation into SQL workloads (primarily join-and-aggregate tensor contractions).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"presentation at ACM SIGMOD 2027 and publication in the Proceedings of the ACM on Management of Data (PACMMOD)","evidence":"Accepted for presentation at ACM SIGMOD 2027 and publication in the Proceedings of the ACM on Management of Data (PACMMOD). This arXiv version is an extended technical report that includes the complete appendix","evidenceUrl":"https://arxiv.org/abs/2607.29134","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"presentation at ACM SIGMOD 2027 and publication in the Proceedings of the ACM on Management of Data (PACMMOD)","reviewStatus":"accepted","decisionRaw":"Accepted for presentation at ACM SIGMOD 2027 and publication in the Proceedings of the ACM on Management of Data (PACMMOD). This arXiv version is an extended technical report that includes the complete appendix","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.29134","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted for presentation at ACM SIGMOD 2027 and publication in the Proceedings of the ACM on Management of Data (PACMMOD). This arXiv version is an extended technical report that includes the complete appendix","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_edf5904501cf14de","familyId":"catalog_family_edf5904501cf14de","name":"InfiniteBench/En.MC","oneLine":"InfiniteBench English Multiple Choice variant - first LLM benchmark featuring average data length surpassing 100K tokens for evaluating long-context capabilities with 12 tasks spanning diverse domains","description":"InfiniteBench English Multiple Choice variant - first LLM benchmark featuring average data length surpassing 100K tokens for evaluating long-context capabilities with 12 tasks spanning diverse domains","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/infinitebench-en.mc","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_edf5904501cf14de"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/infinitebench-en.mc"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"infinitebench-en.mc","url":"https://llm-stats.com/benchmarks/infinitebench-en.mc","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_5f09a49df1dd41c3","familyId":"catalog_family_5f09a49df1dd41c3","name":"InfiniteBench/En.QA","oneLine":"InfiniteBench English Question Answering variant - first LLM benchmark featuring average data length surpassing 100K tokens for evaluating long-context capabilities with 12 tasks spanning diverse domains","description":"InfiniteBench English Question Answering variant - first LLM benchmark featuring average data length surpassing 100K tokens for evaluating long-context capabilities with 12 tasks spanning diverse domains","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/infinitebench-en.qa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5f09a49df1dd41c3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/infinitebench-en.qa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"infinitebench-en.qa","url":"https://llm-stats.com/benchmarks/infinitebench-en.qa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_154f69cfb551a849","familyId":"catalog_family_154f69cfb551a849","name":"InfographicsQA","oneLine":"InfographicVQA dataset with 5,485 infographic images and over 30,000 questions requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills","description":"InfographicVQA dataset with 5,485 infographic images and over 30,000 questions requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/infographicsqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_154f69cfb551a849"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/infographicsqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"infographicsqa","url":"https://llm-stats.com/benchmarks/infographicsqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_infoops-bench_c0a89e01","familyId":"bmf_700e7acd1992","name":"InfoOps Bench","oneLine":"InfoOps Bench is an active, constantly updated benchmark measuring the integrity of frontier language models against co-optation for information operations. It uses real examples from a live monitoring pipeline and tests 17 models from 8 providers.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.28503","pdf":"https://arxiv.org/pdf/2607.28503","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.28503"},"evidence":{"snippet":"In this paper we present an active, constantly updated AI benchmark which measures the integrity of frontier language models against being co-opted for use by authoritarian state \"information operations\": intentional, coordinated activities by one state to influence public opinion and information ecosystems in another state.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.28503"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"InfoOps Bench is an active, constantly updated benchmark measuring the integrity of frontier language models against co-optation for information operations. It uses real examples from a live monitoring pipeline and tests 17 models from 8 providers.","whyItMatters":"This benchmark addresses a novel safety concern: whether models can be co-opted for state-sponsored information operations. It could drive improvements in model integrity and safety.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"da7894add9a7d0713e21a23f26af308f20ff76880cec5a18ae35ce6dae497909"},"motivation":"In this paper we present an active, constantly updated AI benchmark which measures the integrity of frontier language models against being co-opted for use by authoritarian state \"information operations\": intentional, coordinated activities by one state to influence public opinion and information ecosystems in another state.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.28503","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_0280788a4587ff1c","familyId":"catalog_family_0280788a4587ff1c","name":"InfoVQA","oneLine":"InfoVQA dataset with 30,000 questions and 5,000 infographic images requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills","description":"InfoVQA dataset with 30,000 questions and 5,000 infographic images requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/infovqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0280788a4587ff1c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/infovqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"infovqa","url":"https://llm-stats.com/benchmarks/infovqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","vision"],"catalogModelCount":10,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_2c3e90c7ee7e7bbd","familyId":"catalog_family_2c3e90c7ee7e7bbd","name":"InfoVQAtest","oneLine":"InfoVQA test set with infographic images requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills","description":"InfoVQA test set with infographic images requiring joint reasoning over document layout, textual content, graphical elements, and data visualizations with elementary reasoning and arithmetic skills","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/infovqatest","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2c3e90c7ee7e7bbd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/infovqatest"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"infovqatest","url":"https://llm-stats.com/benchmarks/infovqatest","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","vision"],"catalogModelCount":12,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_infrabench_aa520bfd","familyId":"bmf_d9f6fa305b39","name":"InfraBench","oneLine":"InfraBench evaluates AI agents on realistic infrastructure tasks across the system stack and lifecycle, with fine-grained per-check risk assessment and a public leaderboard.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.11234","pdf":"https://arxiv.org/pdf/2608.11234","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.11234"},"evidence":{"snippet":"We present InfraBench, a benchmark suite for evaluating AI agents on realistic infrastructure tasks across the full system stack and full operational lifecycle with fine-grained risk assessment.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11234"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"InfraBench evaluates AI agents on realistic infrastructure tasks across the system stack and lifecycle, with fine-grained per-check risk assessment and a public leaderboard.","whyItMatters":"Addresses the complexity of infrastructure management by providing a benchmark for assessing agent performance across risk dimensions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"aeed5bfd74ed096059e6b63aa447c3d34d752c0caab46918b57d07ccd2d5a67b"},"motivation":"Managing modern computing infrastructure has become a steadily harder problem due to the ever-increasing complexity.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11234","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ins-actbench_3f617677","familyId":"bmf_362d5d3241c5","name":"INS-ActBench","oneLine":"INS-ActBench evaluates actuarial capability in LLMs across knowledge, case reasoning, and tool-based practice with 12,050 tasks from public exams.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.24273","pdf":"https://arxiv.org/pdf/2607.24273","project":null,"code":"https://github.com/FDU-INS/INS-ActBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.24273"},"evidence":{"snippet":"We introduce \\textbf{INS-ActBench}, a comprehensive benchmark for evaluating professional actuarial capability in LLMs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24273"},"ranking":{"90d":{"score":28,"rank":272,"coverage":0.55,"confidence":"Low"}},"description":"INS-ActBench evaluates actuarial capability in LLMs across knowledge, case reasoning, and tool-based practice with 12,050 tasks from public exams.","whyItMatters":"Provides a reproducible foundation for assessing professional actuarial assistance, revealing capability gaps in case reasoning and tool use.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"de113beb623652cc29500c3ef9bc30ad7642c83909ce4dc60e633f626b77e794"},"motivation":"Large Language Models (LLMs) have shown strong potential in financial reasoning, but existing benchmarks often evaluate domain knowledge, numerical reasoning, long-context understanding, and tool use in separate settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24273","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FDU-INS","organizationType":"academic-lab","sourceUrl":"https://github.com/FDU-INS/INS-ActBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_inspire_4ebdcda6","familyId":"bmf_1cb6508098e9","name":"INSPIRE","oneLine":"INSPIRE evaluates instruction-aware speech retrieval, where natural-language instructions specify relevance criteria including semantic content, speaker identity, speaking style, environmental sounds, and combinations. The benchmark includes a fixed dataset and protocol for evaluating retrieval systems.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.SD"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.16203","pdf":"https://arxiv.org/pdf/2608.16203","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.16203"},"evidence":{"snippet":"We introduce INSPIRE, the first benchmark for instruction-aware speech retrieval, in which natural-language instructions dynamically specify relevance criteria, including semantic content, speaker identity, speaking style, environmental sounds, and their combinations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.16203"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"INSPIRE evaluates instruction-aware speech retrieval, where natural-language instructions specify relevance criteria including semantic content, speaker identity, speaking style, environmental sounds, and combinations. The benchmark includes a fixed dataset and protocol for evaluating retrieval systems.","whyItMatters":"Speech retrieval systems currently rely on fixed similarity matching and cannot adapt to diverse user intents. INSPIRE provides a standardized evaluation to compare methods across different retrieval intents, highlighting gaps in handling both semantic and paralinguistic attributes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2cd70d638e1a860164b2f8e94057b54e84008f39071b41a08f668cb09fda23f0"},"motivation":"Existing speech retrieval systems rely on fixed similarity matching and cannot adapt to diverse user intents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.16203","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_d1a33ae8e098bd46","familyId":"catalog_family_d1a33ae8e098bd46","name":"Instruct HumanEval","oneLine":"Instruction-based variant of HumanEval benchmark for evaluating large language models' code generation capabilities with functional correctness using pass@k metric on programming problems","description":"Instruction-based variant of HumanEval benchmark for evaluating large language models' code generation capabilities with functional correctness using pass@k metric on programming problems","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/instruct-humaneval","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d1a33ae8e098bd46"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/instruct-humaneval"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"instruct-humaneval","url":"https://llm-stats.com/benchmarks/instruct-humaneval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_instructmove_3a1c4064","familyId":"bmf_a776350d9ba4","name":"InstructMove","oneLine":"Evaluates instruction-following in robot manipulation through pick-and-place tasks requiring language grounding across category, attribute, spatial, and compositional skills.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.22990v1","pdf":"https://arxiv.org/pdf/2608.22990v1","project":null,"code":"https://github.com/HorizonRobotics/RoboOrchardSim","data":null,"hfPaper":null},"evidence":{"snippet":"We introduce InstructMove, a text-indispensable benchmark for instruction-following manipulation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":11,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22990"},"ranking":{"today":{"score":57,"rank":1,"coverage":0.7,"confidence":"Medium"},"30d":{"score":67,"rank":35,"coverage":0.55,"confidence":"Low"},"90d":{"score":60,"rank":130,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates instruction-following in robot manipulation through pick-and-place tasks requiring language grounding across category, attribute, spatial, and compositional skills.","whyItMatters":"Provides a controlled testbed to diagnose visual shortcut learning in vision-language-action policies, improving reliability of language-conditioned manipulation.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"6a7dd22fb166ed396df4583c685df542a882c1c9957ae96efdb23db062f2eeb4"},"motivation":"Vision-language-action (VLA) models have made general-purpose robot manipulation increasingly plausible by conditioning robot actions on natural-language instructions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally introduced with a clear train-eval protocol, held-out tasks, and public code, enabling reuse and comparison.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce InstructMove, a text-indispensable benchmark for instruction-following manipulation."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22990v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":62,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets an important limitation in robotics instruction following and offers simulation assets with public code, likely to attract attention from embodied AI researchers."},"evaluationMode":"public_reusable","publishers":[{"name":"Horizon Robotics","organizationType":"company-research-lab","sourceUrl":"https://github.com/HorizonRobotics/RoboOrchardSim","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_insufficiencybench_debc0b74","familyId":"bmf_1d2d1cba14d5","name":"InsufficiencyBench","oneLine":"Evaluates whether LLMs recognize legally material missing information in underspecified queries and refrain from premature conclusions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.20220","pdf":"https://arxiv.org/pdf/2608.20220","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce InsufficiencyBench, the first legal benchmark targeting query-side insufficiency: whether a model recognizes when a query lacks legally material information, identifies what is missing, and refrains from premature conclusions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.20220"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates whether LLMs recognize legally material missing information in underspecified queries and refrain from premature conclusions.","whyItMatters":"Addresses the practical gap of query-side insufficiency in legal AI, with attorney-annotated items across domains and jurisdictions.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"13a0a40fa2d3268541a631cae2cfb4a97e3598bc757836b5f61be30d7e931e5e"},"motivation":"Legal AI systems are increasingly used to answer legal questions, yet existing benchmarks assume queries arrive fully specified.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally named, has 202 attorney-annotated items with an 8-category taxonomy and evaluation metrics, and received recognition at an AI4Law workshop.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce InsufficiencyBench, the first legal benchmark targeting query-side insufficiency"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.20220","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:50.392548Z"},"attentionForecast":{"score":52,"confidence":"Low","horizon":"7d","reason":"InsufficiencyBench has substantial expert annotation and an award, but the source lacks an explicit release statement beyond the paper."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_int-bench_bde8cb1b","familyId":"bmf_16d5485f629c","name":"Int-Bench","oneLine":"Int-Bench is a simulation-based evaluation for LLM intervention behavior in tutoring. It simulates a student solving problems across code debugging, mathematics, and brain teasers, with a teacher deciding whether, when, and how to intervene, measuring frequency, timing, and impact on task success and generalization.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.21306","pdf":"https://arxiv.org/pdf/2607.21306","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.21306"},"evidence":{"snippet":"Here, we introduce Int-Bench, a simulation-based benchmark for evaluating LLM interventions during learning.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.21306"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Int-Bench is a simulation-based evaluation for LLM intervention behavior in tutoring. It simulates a student solving problems across code debugging, mathematics, and brain teasers, with a teacher deciding whether, when, and how to intervene, measuring frequency, timing, and impact on task success and generalization.","whyItMatters":"The evaluation gap is that AI assistants may over-assist, hindering learning. This benchmark aims to quantify intervention timing and content, providing a method to compare models on supportive versus answer-giving behavior, which is critical for educational AI.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bda1c89cb908767226d3a31994d3965f6b3a84b8e8244752aeea4277fea5f40d"},"motivation":"Large language models (LLMs) are increasingly used as tutors and thought partners, helping users reason through problems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.21306","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_integritybench_9bf5a285","familyId":"bmf_6062d4bd1c8f","name":"IntegrityBench","oneLine":"IntegrityBench evaluates language models on research integrity tasks, including misconduct classification, ethical action reasoning, and artifact-grounded decision making, across 36 paired tasks with a 5-level pressure protocol spanning multiple domains and research stages.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12345","pdf":"https://arxiv.org/pdf/2608.12345","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12345"},"evidence":{"snippet":"We introduce IntegrityBench, a benchmark evaluating misconduct classification, ethical action reasoning and artifact-grounded decision making across 36 paired tasks under a 5-level implicit-explicit pressure protocol spanning 3 domains and 4 research stages.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12345"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"IntegrityBench evaluates language models on research integrity tasks, including misconduct classification, ethical action reasoning, and artifact-grounded decision making, across 36 paired tasks with a 5-level pressure protocol spanning multiple domains and research stages.","whyItMatters":"As language models are used as co-scientists, measuring their integrity under pressure is critical. This benchmark could inform deployment decisions and identify risks of facilitating misconduct or eroding trust, but the evaluation method and reproducibility are not yet specified.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cb40d8ed1c2e5ef70781fc6d891ecb8e293308e57da58fa99f9ff7db981e39a2"},"motivation":"Language models are increasingly deployed as co-scientists, yet their ability to uphold research integrity under institutional pressure remains unmeasured.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12345","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_intentbench-prime_088c7112","familyId":"bmf_0161280c13b7","name":"Intentbench-Prime","oneLine":"Intentbench-Prime is a cleaned subset of the IntentBench audio-visual social QA benchmark with broken and trivially answerable questions removed.","area":"Multimodal","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":0.75,"links":{"report":"https://arxiv.org/abs/2608.13239","pdf":"https://arxiv.org/pdf/2608.13239","project":null,"code":"https://github.com/koenv759/VanillaSFT","data":null,"hfPaper":null},"evidence":{"snippet":"We remove the affected questions and release Intentbench-Prime.","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13239"},"ranking":{"30d":{"score":23,"rank":151,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":355,"coverage":0.55,"confidence":"Low"}},"description":"Intentbench-Prime is a cleaned subset of the IntentBench audio-visual social QA benchmark with broken and trivially answerable questions removed.","whyItMatters":"Provides a higher-quality evaluation set for social audio-visual question answering, reducing noise and enabling fairer comparison of reasoning methods.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"89db204ab0f2a5a342d405d3f3d730d1c24946e46f1b5d6e63a414fdeb73c141"},"motivation":"Training Multimodal Large Language Models for audio-visual social understanding is a crucial step toward embodied social intelligence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is explicitly released and publicly available via a companion repository with a provided scorer and evaluation protocol.","canonicalNameSource":"abstract","canonicalNameEvidence":"We remove the affected questions and release Intentbench-Prime."},"publication":{"status":"acceptance_claimed","venue":"HCMIW ECCV workshop","evidence":"Accepted at HCMIW ECCV workshop. Code available here: https://github.com/koenv759/VanillaSFT","evidenceUrl":"https://arxiv.org/abs/2608.13239","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-19T11:05:13.395391Z"},"venueAttempts":[{"venueName":"HCMIW ECCV workshop","reviewStatus":"accepted","decisionRaw":"Accepted at HCMIW ECCV workshop. Code available here: https://github.com/koenv759/VanillaSFT","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.13239","observedAt":"2026-08-19T11:05:13.395391Z","rawValue":"Accepted at HCMIW ECCV workshop. Code available here: https://github.com/koenv759/VanillaSFT","level":"author-claim"}]}],"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"The benchmark's release alongside surprising findings and accessible artifacts is likely to draw moderate attention from the multimodal reasoning community."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_intentionnav_63acc227","familyId":"bmf_5dd62d05b1f1","name":"IntentionNav","oneLine":"IntentionNav evaluates active object search from implicit human instructions in 176 Isaac Sim scenes. Episodes provide free-text intent, RGB-D observations, and pose, with the target object withheld. The benchmark includes 500 intents over 64 categories and four intent modes.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.23187","pdf":"https://arxiv.org/pdf/2605.23187","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.23187"},"evidence":{"snippet":"We study this setting as intent-driven object navigation and introduce IntentionNav, a diagnostic benchmark for active object search from implicit human instructions.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.23187"},"ranking":{},"description":"IntentionNav evaluates active object search from implicit human instructions in 176 Isaac Sim scenes. Episodes provide free-text intent, RGB-D observations, and pose, with the target object withheld. The benchmark includes 500 intents over 64 categories and four intent modes.","whyItMatters":"Addresses the gap in object navigation where agents must infer targets from indirect human intent, showing persistent bottlenecks in target selection and terminal localization.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5b88d5dd64953bce8f1aa1d7a5e40bce0026236812ec80b12c0b99a1c2e1165c"},"motivation":"Existing object navigation benchmarks usually tell an embodied agent which object category to find, such as microwave or chair.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.23187","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_inter-x-a-comprehensive-benchmark-for-mult_a4cc7946","familyId":"bmf_183ef236693e","name":"Inter-X++","oneLine":"Large-scale benchmark with 11,388 interaction sequences for multimodal human-human interaction analysis, covering four task categories in generation and perception.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.20312","pdf":"https://arxiv.org/pdf/2608.20312","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To systematically address these bottlenecks, we present Inter-X++, a comprehensive and large-scale benchmark designed to empower versatile HHI analysis.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.20312"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Large-scale benchmark with 11,388 interaction sequences for multimodal human-human interaction analysis, covering four task categories in generation and perception.","whyItMatters":"Provides standardized representations and evaluation protocols for human-human interaction, enabling fair comparison across generative and perceptive models.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"6dba21870bf0dea4a7b5a3a297ba3d15dc2c224ba90b644f4ea5bdff86f36654"},"motivation":"The capability to perceive and synthesize human-human interactions is fundamental to developing intelligent digital human systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The paper presents the benchmark with unified task definitions and evaluation protocols, and the dataset is intended for public use, though direct artifact links are not provided.","canonicalNameSource":"paper_title","canonicalNameEvidence":"Inter-X++: A Comprehensive Benchmark for Multimodal Human-Human Interaction Analysis"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.20312","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:50.392548Z"},"attentionForecast":{"score":60,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses a gap in multimodal human interaction data and introduces a unified framework, likely attracting attention from digital human and embodied AI researchers."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_interflopbench_d649e042","familyId":"bmf_601ef5f2c623","name":"InterFLOPBench","oneLine":"InterFLOPBench is a benchmark of 90 C kernels and 1,130 test samples for evaluating LLMs on floating-point error classification across six categories: cancellation, comparison, division by zero, overflow, underflow, and NaN.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.31308","pdf":"https://arxiv.org/pdf/2606.31308","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.31308"},"evidence":{"snippet":"We introduce InterFLOPBench, a benchmark of 90 C kernels with 1 130 test samples designed to evaluate LLMs across six categories of floating-point error: cancellation, comparison, division by zero, overflow, underflow and NaN, compared across 14 LLMs.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31308"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"InterFLOPBench is a benchmark of 90 C kernels and 1,130 test samples for evaluating LLMs on floating-point error classification across six categories: cancellation, comparison, division by zero, overflow, underflow, and NaN.","whyItMatters":"Provides a targeted evaluation for LLM capabilities in static floating-point error detection, a niche but important area. Enables comparison across models and error types.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b18394eee69131223a87b869faa349657c075afa90557d84f88fb0cd3a002a4e"},"motivation":"This paper investigates the capability of Large Language Models (LLMs) to detect and classify floating-point errors statically in software code.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31308","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_900df6bdc7f9459d","familyId":"catalog_family_900df6bdc7f9459d","name":"InterGPS","oneLine":"Interpretable Geometry Problem Solver (Inter-GPS) with Geometry3K dataset of 3,002 geometry problems with dense annotation in formal language using theorem knowledge and symbolic reasoning","description":"Interpretable Geometry Problem Solver (Inter-GPS) with Geometry3K dataset of 3,002 geometry problems with dense annotation in formal language using theorem knowledge and symbolic reasoning","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Spatial Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/intergps","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_900df6bdc7f9459d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/intergps"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"intergps","url":"https://llm-stats.com/benchmarks/intergps","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","spatial reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_2fbf5fa976d9cd7d","familyId":"catalog_family_2fbf5fa976d9cd7d","name":"Internal API instruction following (hard)","oneLine":"Internal API instruction following (hard) benchmark - specific documentation not found in official sources","description":"Internal API instruction following (hard) benchmark - specific documentation not found in official sources","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Structured Output","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/internal-api-instruction-following-(hard)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2fbf5fa976d9cd7d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/internal-api-instruction-following-(hard)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"internal-api-instruction-following-(hard)","url":"https://llm-stats.com/benchmarks/internal-api-instruction-following-(hard)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["structured output","general"],"catalogModelCount":7,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general"},{"id":"catalog_228095751750f019","familyId":"catalog_family_228095751750f019","name":"Internal Research Debugging Evaluation","oneLine":"The Internal Research Debugging Evaluation measures whether models can debug 41 real bugs from internal OpenAI research experiments (plus alignment-auditing tasks), where the original solutions took experienced researchers hours to days. Passing corresponds to providing assistance that would unblock the user, including partial root-cause explanations or fixes.","description":"The Internal Research Debugging Evaluation measures whether models can debug 41 real bugs from internal OpenAI research experiments (plus alignment-auditing tasks), where the original solutions took experienced researchers hours to days. Passing corresponds to providing assistance that would unblock the user, including partial root-cause explanations or fixes.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/internal-research-debugging-evaluation","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_228095751750f019"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/internal-research-debugging-evaluation"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"internal-research-debugging-evaluation","url":"https://llm-stats.com/benchmarks/internal-research-debugging-evaluation","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","code"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_inverse-turing-bench_7691d060","familyId":"bmf_3c53c6fe0c9e","name":"Inverse Turing Bench","oneLine":"The benchmark evaluates language models on distinguishing human-only vs. human-AI multi-turn dialogues, using paired transcripts and accuracy as the metric.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.21844","pdf":"https://arxiv.org/pdf/2606.21844","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.21844"},"evidence":{"snippet":"We present Inverse Turing Bench, a benchmark that evaluates LLMs and other models on their ability to differentiate humans and AI in multi-turn text.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.21844"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"The benchmark evaluates language models on distinguishing human-only vs. human-AI multi-turn dialogues, using paired transcripts and accuracy as the metric.","whyItMatters":"It addresses the practical need for reliable human-AI differentiation in online spaces, with implications for trust and safety. The benchmark may help compare detection approaches, though its current form primarily supports the paper's findings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9b22c4ab48773d9c9bf76944e2a21876b67978012142db356ca68d7dc5fb4507"},"motivation":"As AI systems integrate into online spaces, differentiating them from humans in conversations is increasingly important.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.21844","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_investlogicbench_44a26030","familyId":"bmf_a9cc3f68edd1","name":"InvestLogicBench","oneLine":"InvestLogicBench evaluates large language models on personalized investment decision-making using 201,247 documented decisions from 151 real-world investors. Each episode traces investor profile, market events, reasoning, decision, and outcome. Tasks include comprehension, profile-conditioned generation, and end-to-end replay.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.06108","pdf":"https://arxiv.org/pdf/2608.06108","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.06108"},"evidence":{"snippet":"We introduce \\textsc{InvestLogicBench}, a process-native benchmark containing 201,247 documented decisions from 151 real-world investors.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06108"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"InvestLogicBench evaluates large language models on personalized investment decision-making using 201,247 documented decisions from 151 real-world investors. Each episode traces investor profile, market events, reasoning, decision, and outcome. Tasks include comprehension, profile-conditioned generation, and end-to-end replay.","whyItMatters":"Existing financial LLM evaluations rely on static QA or terminal profit, which fail to reveal whether actions are profile-consistent or grounded in events. This benchmark targets a gap by assessing process quality and grounding in personalized, consequential settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e16a727c3ab15df9f6562cedff4bbc09cffe4a48865876a5cd79128e01041677"},"motivation":"Investment competence is inherently personalized: the same market evidence can justify different actions for investors with different goals, horizons, portfolios, and risk boundaries.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06108","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_investphilbench_547d0c00","familyId":"bmf_2ef9232fbd60","name":"InvestPhilBench","oneLine":"InvestPhilBench is a multi-layer benchmark for evaluating procedural reasoning in investment philosophy, spanning eight cognitive tiers from principle identification to novel framework extrapolation. It includes principle cards, decision-framework cards, and QA questions, with an automated scoring pipeline (BASP) and five algorithmic metrics.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.25984","pdf":"https://arxiv.org/pdf/2606.25984","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.25984"},"evidence":{"snippet":"We introduce InvestPhilBench, a multi-layer benchmark spanning eight cognitive tiers, from principle identification (L1) to novel framework extrapolation (L8).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.25984"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"InvestPhilBench is a multi-layer benchmark for evaluating procedural reasoning in investment philosophy, spanning eight cognitive tiers from principle identification to novel framework extrapolation. It includes principle cards, decision-framework cards, and QA questions, with an automated scoring pipeline (BASP) and five algorithmic metrics.","whyItMatters":"Fills the gap in testing whether LLMs can accurately reconstruct and apply expert procedural decision frameworks. Provides a reproducible method for scoring procedural reasoning, with a metric (GRA) that exposes deficits hidden by composite scores.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"40867a345cfe01336a8ae3495be03bad7db765cf78bb971da89c034bca5b94ef"},"motivation":"Large language models are increasingly deployed as investment research assistants, yet no benchmark tests whether they can accurately reconstruct and apply the specific procedural decision frameworks of expert investors.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.25984","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_b2b3a4a1879c2312","familyId":"catalog_family_b2b3a4a1879c2312","name":"IOI","oneLine":"Based on the International Olympiad in Informatics","description":"Based on the International Olympiad in Informatics","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/ioi","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b2b3a4a1879c2312"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsioi"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsIoi","url":"https://benchlm.ai/benchmarks/valsioi","paperUrl":"https://www.vals.ai/benchmarks/ioi","year":"2026","fullName":"Vals IOI","format":"Accuracy score","tasks":"International Olympiad in Informatics-style programming tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_iosworld_171f5010","familyId":"bmf_5df7fedfa2e7","name":"iOSWorld","oneLine":"iOSWorld evaluates phone agents on interactive tasks within a native iOS simulator. It includes 26 custom iOS apps with connected personal data, 133 tasks across single-app, multi-app, and memory/personalization categories. Agents are evaluated with rubric-based scoring.","area":"Agents & Tool Use","applicationDomains":["Consumer & Productivity"],"primaryDomain":"Consumer & Productivity","industrySectors":["Consumer Technology"],"capabilities":["Computer use","Cross-app planning","Memory & personalization"],"topics":["Agents","Mobile","Personalization"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09764","pdf":"https://arxiv.org/pdf/2606.09764","project":"https://iosworld.io/","code":"https://github.com/ljang0/iOSWorld","data":null,"hfPaper":"https://huggingface.co/papers/2606.09764"},"evidence":{"snippet":"We introduce iOSWorld, the first interactive native iOS simulator benchmark built around a persistent user identity spanning 26 newly built iOS apps.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-reviewed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":4,"hfDailySubmittedAt":"2026-06-18T00:00:00.000Z","githubStars":15,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09764"},"ranking":{"90d":{"score":40,"rank":152,"coverage":0.7,"confidence":"Medium"}},"description":"iOSWorld evaluates phone agents on interactive tasks within a native iOS simulator. It includes 26 custom iOS apps with connected personal data, 133 tasks across single-app, multi-app, and memory/personalization categories. Agents are evaluated with rubric-based scoring.","whyItMatters":"Existing mobile benchmarks lack personalization and interactive evaluation. iOSWorld provides a benchmark with persistent user identity and multi-app tasks, showing significant gaps in multi-app and memory-based performance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a8045efb73c464679ea35312b31b67d30645c825b22ef2dfac3fa1eb6a8ccc91"},"motivation":"A useful phone agent needs to be personally intelligent.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"source-reviewed","reviewedAt":"2026-08-19","sources":["https://arxiv.org/abs/2606.09764","https://iosworld.io/","https://github.com/ljang0/iOSWorld"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09764","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"iOSWorld Team","organizationType":"academic-lab","sourceUrl":"https://github.com/ljang0/iOSWorld","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_e227df5e4828f1a8","familyId":"catalog_family_e227df5e4828f1a8","name":"IPhO 2025","oneLine":"International Physics Olympiad 2025 (theory) comprises all 3 theory problems from the official 2025 IPhO competition. Results are based on blinded human evaluation with guidelines based on the official competition scoring, validated by domain experts.","description":"International Physics Olympiad 2025 (theory) comprises all 3 theory problems from the official 2025 IPhO competition. Results are based on blinded human evaluation with guidelines based on the official competition scoring, validated by domain experts.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Physics","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ipho-2025","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e227df5e4828f1a8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ipho-2025"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ipho-2025","url":"https://llm-stats.com/benchmarks/ipho-2025","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["physics","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_8651e56eb8432f4b","familyId":"catalog_family_8651e56eb8432f4b","name":"IPhO 2025 (Theory)","oneLine":"The three official theory problems from the 2025 International Physics Olympiad, scored with blinded human evaluation.","description":"The three official theory problems from the 2025 International Physics Olympiad, scored with blinded human evaluation.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8651e56eb8432f4b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/ipho2025theory"}],"catalogSources":[{"catalog":"benchlm","sourceId":"ipho2025Theory","url":"https://benchlm.ai/benchmarks/ipho2025theory","paperUrl":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","year":"2026","fullName":"International Physics Olympiad 2025 (Theory)","format":"Physics olympiad theory","tasks":"3 olympiad theory problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_ipibench_3c35c7c4","familyId":"bmf_7ebfbd9fee5c","name":"IPIBench","oneLine":"Benchmark for interactive proactive intelligence of MLLMs under streaming video, covering proactive monitoring, task management, and interleaved requests.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27074","pdf":"https://arxiv.org/pdf/2605.27074","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27074"},"evidence":{"snippet":"To address this gap, we introduce IPIBench, the first benchmark for evaluating Interactive Proactive Intelligence of MLLMs under streaming video settings.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27074"},"ranking":{},"description":"Benchmark for interactive proactive intelligence of MLLMs under streaming video, covering proactive monitoring, task management, and interleaved requests.","whyItMatters":"Existing benchmarks overlook dynamic multi-turn proactive interactions. IPIBench fills this gap for streaming assistant evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9f7cf8d7b8ba4c0e2f0251e8167e42d6910bd50f145473c1950a5047f9933658"},"motivation":"Recent multimodal large language models (MLLMs) achieve strong performance on reactive question answering, but real-world streaming assistants require proactive reasoning over continuous visual inputs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27074","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_ir275k_3c0945b4","familyId":"bmf_291981a5e03d","name":"IR275K","oneLine":"A benchmark for infrared multi-frame super-resolution containing 594 video sequences and 275,196 frames with fixed sequence-level splits and a reproducible x4 evaluation protocol.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.22380","pdf":"https://arxiv.org/pdf/2607.22380","project":null,"code":"https://github.com/InfraRecon7/IR275K","data":null,"hfPaper":"https://huggingface.co/papers/2607.22380"},"evidence":{"snippet":"We introduce IR275K, a curated benchmark containing 594 infrared video sequences and 275,196 frames.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":22,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.22380"},"ranking":{"90d":{"score":46,"rank":98,"coverage":0.55,"confidence":"Low"}},"description":"A benchmark for infrared multi-frame super-resolution containing 594 video sequences and 275,196 frames with fixed sequence-level splits and a reproducible x4 evaluation protocol.","whyItMatters":"Provides a standardized evaluation resource for accuracy-efficiency trade-offs in infrared MFSR, a domain with fragmented evaluation practices.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"efe421530bdc246cb740bd0f3720a070429d1f4d7280987507d8973b28098e34"},"motivation":"Efficient processing is becoming increasingly important in infrared remote sensing, where satellite constellations produce large volumes of observations under constrained detector resolution, power, and downlink bandwidth.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.22380","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_irts-toolbench_94993a2c","familyId":"bmf_cce157d7294f","name":"IRTS-ToolBench","oneLine":"IRTS-ToolBench is a benchmark of 1,700 questions across 10 task types and 13 domains for evaluating irregular univariate time-series question answering, with standardized inputs and a reproducible evaluation protocol.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15107","pdf":"https://arxiv.org/pdf/2606.15107","project":null,"code":"https://github.com/SanhornC/IRTS-ToolBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.15107"},"evidence":{"snippet":"To bridge this gap, we introduce IRTS-ToolBench, a benchmark of 1,700 questions spanning 10 task types across 13 domains.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15107"},"ranking":{"90d":{"score":28,"rank":281,"coverage":0.55,"confidence":"Low"}},"description":"IRTS-ToolBench is a benchmark of 1,700 questions across 10 task types and 13 domains for evaluating irregular univariate time-series question answering, with standardized inputs and a reproducible evaluation protocol.","whyItMatters":"Existing TSQA benchmarks assume regular sampling, leaving a gap for real-world irregular data. This benchmark provides standardized evaluation for LLMs and AI agents on irregular time series, with golden tool sets for tool-selection analysis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7f0189ab569b5b0bf93070aa3fde64a64e744f7b6446d8edd1949923a10a6465"},"motivation":"Time series data in real-world deployments is overwhelmingly irregular.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15107","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ishigaki-ids-bench_d58a0879","familyId":"bmf_4b152674589d","name":"Ishigaki-IDS-Bench","oneLine":"A benchmark for generating Information Delivery Specification (IDS) XML from BIM information requirements, with 166 examples in English and Japanese. It evaluates LLMs on formal validity via IDSAuditTool and content fidelity via facet-level macro-F1 against gold IDS files.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.22079","pdf":"https://arxiv.org/pdf/2605.22079","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.22079"},"evidence":{"snippet":"We present Ishigaki-IDS-Bench, the first publicly released benchmark for IDS generation from BIM information requirements.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.22079"},"ranking":{},"description":"A benchmark for generating Information Delivery Specification (IDS) XML from BIM information requirements, with 166 examples in English and Japanese. It evaluates LLMs on formal validity via IDSAuditTool and content fidelity via facet-level macro-F1 against gold IDS files.","whyItMatters":"IDS authoring requires domain expertise and vocabulary conformance; this benchmark quantifies LLM performance on a structured generation task where output must satisfy external validation tools, filling a gap for capability assessment in BIM-specific language models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c27197e53cf8db2dad459e76339bd492abf8bcdc7e06d7a1555d7d6cf7f0931b"},"motivation":"Building Information Modeling (BIM) projects increasingly use Information Delivery Specification (IDS) to formalize information requirements in a machine-checkable XML format.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"CIKM '26 (35th ACM International Conference on Information and Knowledge Management), Rome, Italy, November 2026","evidence":"5 pages, 2 figures. Accepted at CIKM '26 (35th ACM International Conference on Information and Knowledge Management), Rome, Italy, November 2026; resource track, oral presentation. ACM DOI: 10.1145/3799682.3840206. Benchmark data on Hugging Face and evaluation code on Zenodo","evidenceUrl":"https://arxiv.org/abs/2605.22079","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"CIKM '26 (35th ACM International Conference on Information and Knowledge Management), Rome, Italy, November 2026","reviewStatus":"accepted","decisionRaw":"5 pages, 2 figures. Accepted at CIKM '26 (35th ACM International Conference on Information and Knowledge Management), Rome, Italy, November 2026; resource track, oral presentation. ACM DOI: 10.1145/3799682.3840206. Benchmark data on Hugging Face and evaluation code on Zenodo","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2605.22079","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"5 pages, 2 figures. Accepted at CIKM '26 (35th ACM International Conference on Information and Knowledge Management), Rome, Italy, November 2026; resource track, oral presentation. ACM DOI: 10.1145/3799682.3840206. Benchmark data on Hugging Face and evaluation code on Zenodo","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_isosci_576d9a68","familyId":"bmf_c5ac3d50efe8","name":"IsoSci","oneLine":"IsoSci evaluates reasoning vs. knowledge retrieval in LLMs using isomorphic cross-domain problem pairs, enabling controlled attribution of reasoning gains.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Information retrieval","Factuality"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.01431","pdf":"https://arxiv.org/pdf/2607.01431","project":null,"code":null,"data":"https://huggingface.co/datasets/isosci/isosci","hfPaper":"https://huggingface.co/papers/2607.01431"},"evidence":{"snippet":"We introduce ISOSCI, a benchmark of isomorphic cross-domain science problem pairs that separates reasoning ability from domain knowledge retrieval in LLM evaluation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":33,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2607.01431"},"ranking":{"90d":{"score":44,"rank":119,"coverage":0.3,"confidence":"Low","datasetDownloadRank":65,"datasetRankPopulation":66}},"description":"IsoSci evaluates reasoning vs. knowledge retrieval in LLMs using isomorphic cross-domain problem pairs, enabling controlled attribution of reasoning gains.","whyItMatters":"Provides a method to separate reasoning ability from knowledge, challenging assumptions about chain-of-thought and guiding model evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"76a48fd05695f3d246651a97e75b6d54882ec2995b71eb870668244065323ede"},"motivation":"We introduce ISOSCI, a benchmark of isomorphic cross-domain science problem pairs that separates reasoning ability from domain knowledge retrieval in LLM evaluation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01431","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_itpeval_0f85127f","familyId":"bmf_edcd96daf17a","name":"ITPEval","oneLine":"ITPEval evaluates automated formal proof translation across four interactive theorem provers (Lean 4, Rocq, Isabelle, HOL Light), with 1,560 source files and 6,848 theorems, covering statement and proof translation on 12 directed pairs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19407","pdf":"https://arxiv.org/pdf/2607.19407","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19407"},"evidence":{"snippet":"We present ITPEval, the first benchmark for evaluating automated formal proof translation across four major ITPs (Lean 4, Rocq, Isabelle, and HOL Light), spanning two distinct logical foundations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19407"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ITPEval evaluates automated formal proof translation across four interactive theorem provers (Lean 4, Rocq, Isabelle, HOL Light), with 1,560 source files and 6,848 theorems, covering statement and proof translation on 12 directed pairs.","whyItMatters":"ITPEval addresses the lack of a unified benchmark for cross-prover formal proof translation, providing a standardized evaluation of a key capability for automated reasoning and verified software portability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d50cedcc907e1b100a284c56cc9839dbb3cbb7076415c193f377570d249016d1"},"motivation":"Formal theorem proving has emerged as a frontier challenge for machine learning, yet the ecosystem is fragmented: proofs remain siloed across incompatible systems, limiting both training data for learning-based provers and the portability of verified results.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19407","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ITPEval team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.19407","role":"benchmark-publisher"}],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_janus_33d6bb69","familyId":"bmf_850ca8e0ec3f","name":"Janus","oneLine":"JANUS is a benchmark with 160 scenarios across 8 domains, each providing a fixed pool of favorable and adverse facts and paired neutral and goal-directed prompts, to evaluate fact-grounded, goal-conditioned pragmatic distortion in LLM outputs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.10852","pdf":"https://arxiv.org/pdf/2606.10852","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.10852"},"evidence":{"snippet":"We introduce JANUS, a benchmark for measuring goal-conditioned pragmatic distortion in fact-grounded LLM outputs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.10852"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"JANUS is a benchmark with 160 scenarios across 8 domains, each providing a fixed pool of favorable and adverse facts and paired neutral and goal-directed prompts, to evaluate fact-grounded, goal-conditioned pragmatic distortion in LLM outputs.","whyItMatters":"Existing benchmarks primarily detect direct deception, missing subtler misleading communication that stays factually accurate. JANUS measures whether LLMs distort net impressions when incentivized, offering a more practical assessment of risks in real-world applications where selective presentation of facts can mislead stakeholders.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c2ad0b2a215ac65f5dc2350a20137a6c20035787766987448697827a7a6a842c"},"motivation":"LLM deception is often evaluated through direct markers such as fabricated claims, explicit lies, or strategic concealment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.10852","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"JANUS Benchmark Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.10852","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_jarvisbench_2366ad17","familyId":"bmf_4e23273b5c08","name":"JarvisBench","oneLine":"JarvisBench measures mediation in long-horizon agent workflows with two tracks: agent-collaboration and user-interaction. Built on WildClaw tasks and a reference Jarvis prototype.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.16610","pdf":"https://arxiv.org/pdf/2607.16610","project":"https://cchen1436.github.io/jarvis","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.16610"},"evidence":{"snippet":"In this work, we introduce JarvisBench, a benchmark for measuring the dual value of mediation in long-horizon agent workflows.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.16610"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"JarvisBench measures mediation in long-horizon agent workflows with two tracks: agent-collaboration and user-interaction. Built on WildClaw tasks and a reference Jarvis prototype.","whyItMatters":"Could fill a gap in evaluating agent-user interaction, but unclear path for reuse.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"14bc45d8d26c66fd5d6d72e756157a1a41832f265b96b59b70f9eb3c49ca135f"},"motivation":"Long-horizon AI agents are becoming increasingly capable, yet their interaction with users remains surprisingly thin.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.16610","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_javavulbench_93d353dc","familyId":"bmf_c5f6848b3970","name":"JavaVulBench","oneLine":"JavaVulBench is a benchmark dataset and evaluation harness for Java vulnerability detection. The dataset includes about 30,600 Java methods spanning 1,740 CVEs and 700+ projects, with method and line labels, publication dates, and five split strategies. The harness provides a unified schema across multiple backends and includes a contamination audit.","area":"Vision & 3D","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.02825","pdf":"https://arxiv.org/pdf/2607.02825","project":"https://www.youtube.com/watch?v=nMTX\\_hqkuoM","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.02825"},"evidence":{"snippet":"We release \\textsc{JavaVulBench}, a benchmark dataset and evaluation harness for Java vulnerability detection.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02825"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"JavaVulBench is a benchmark dataset and evaluation harness for Java vulnerability detection. The dataset includes about 30,600 Java methods spanning 1,740 CVEs and 700+ projects, with method and line labels, publication dates, and five split strategies. The harness provides a unified schema across multiple backends and includes a contamination audit.","whyItMatters":"It addresses the need for realistic and leakage-aware evaluation of vulnerability detection models, offering multiple split strategies and contamination audits to separate genuinely unseen CVEs from memorized ones.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cb72eb271cd2c494f1b7ccfd68ed2000fda3b7e3a003465535675ada1e918fb7"},"motivation":"We release \\textsc{JavaVulBench}, a benchmark dataset and evaluation harness for Java vulnerability detection.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02825","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_jedi_9de96442","familyId":"bmf_17779b11c2c9","name":"JEDI","oneLine":"JEDI is a benchmark suite for the Java Stream API, generated by converting SQL benchmarks into Java benchmarks. It includes both stream-based and imperative query implementations to evaluate performance of parallelization strategies.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.PL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.23543","pdf":"https://arxiv.org/pdf/2605.23543","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.23543"},"evidence":{"snippet":"In this work we present JEDI, a benchmark suite that targets the Stream API.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.23543"},"ranking":{},"description":"JEDI is a benchmark suite for the Java Stream API, generated by converting SQL benchmarks into Java benchmarks. It includes both stream-based and imperative query implementations to evaluate performance of parallelization strategies.","whyItMatters":"There is a lack of benchmarks for the Java Stream API, making it difficult to optimize the API and analyze stream performance. JEDI provides a dedicated suite to guide developers in writing efficient code and researchers in optimizing the API.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a5d2732a73a0ca31ba9936ca344cff1a640d97dba4466c0f908814fb31df80b5"},"motivation":"The Java Stream API aims at increasing developer productivity thanks to an easy-to-read declarative syntax to express computations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"publication_reported","venue":"2026 IEEE/ACM 48th International Conference on Software Engineering (ICSE '26)","evidence":"2026 IEEE/ACM 48th International Conference on Software Engineering (ICSE '26)","evidenceUrl":"https://arxiv.org/abs/2605.23543","source":"arxiv-journal-reference","evidenceLevel":"strong-author-metadata","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publications":[{"venueName":"2026 IEEE/ACM 48th International Conference on Software Engineering (ICSE '26)","publicationStatus":"published","evidence":[{"sourceType":"arxiv-journal-reference","sourceUrl":"https://arxiv.org/abs/2605.23543","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"2026 IEEE/ACM 48th International Conference on Software Engineering (ICSE '26)","level":"strong-author-metadata"}]}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_jeto-bench_f10e93b0","familyId":"bmf_714867b9d42b","name":"JETO-Bench","oneLine":"JETO-Bench is a benchmark of 660 execution time improvement patches (ETIPs) in Java, with 91 manually verified. It is built using JETO-Mine, a configurable tool that mines ETIPs from GitHub repositories and statistically validates improvements.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.31767","pdf":"https://arxiv.org/pdf/2606.31767","project":null,"code":"https://github.com/khesoem/JETO-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.31767"},"evidence":{"snippet":"Using JETO-Mine, we build JETO-Bench, a benchmark of 660 identified and 91 manually verified executable ETIPs from 174 Java repositories, mined from nearly 1.8 million commits across 11 years.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31767"},"ranking":{"90d":{"score":28,"rank":277,"coverage":0.55,"confidence":"Low"}},"description":"JETO-Bench is a benchmark of 660 execution time improvement patches (ETIPs) in Java, with 91 manually verified. It is built using JETO-Mine, a configurable tool that mines ETIPs from GitHub repositories and statistically validates improvements.","whyItMatters":"Provides a reproducible benchmark for performance bug fixing in Java, an area lacking benchmarks. Enables evaluation of patch generation and test generation tools.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"94cc332a0e494acb03efaa68cb0a40813a28fbca709b4e6782b667f0addeab8c"},"motivation":"Automated fixing of performance issues is gaining attention, but existing benchmarks of execution time improvement patches (ETIPs) target Python, C++, or .NET and are fixed datasets that cannot be extended under user-defined configurations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31767","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"khesoem","organizationType":"academic-lab","sourceUrl":"https://github.com/khesoem/JETO-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_jitoma-bench_daae8913","familyId":"bmf_ddb468992b92","name":"JITOMA-Bench","oneLine":"JITOMA-Bench is a suite for long-horizon multi-tasking and multi-step reasoning in robotics, focusing on just-in-time scene graph growth to combat perceptual saturation.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.13245","pdf":"https://arxiv.org/pdf/2607.13245","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.13245"},"evidence":{"snippet":"To evaluate these dynamic capabilities and study perceptual saturation trade-offs, we introduce JITOMA-Bench, a comprehensive suite for long-horizon multi-tasking and complex multi-step reasoning.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.13245"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"JITOMA-Bench is a suite for long-horizon multi-tasking and multi-step reasoning in robotics, focusing on just-in-time scene graph growth to combat perceptual saturation.","whyItMatters":"The benchmark supports the JITOMA framework's evaluation, but its primary purpose is to demonstrate the framework's advantages rather than serve as a standalone comparison benchmark.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d1ba2493643e2e1b62f8b60a9ca016d40196b8418369b28a7d5e85a0d0fc872b"},"motivation":"While 3D Scene Graphs (3DSGs) provide crucial structured representations for embodied agents, conventional Ahead-of-Time, build-everything-then-filter pipelines conflict with the real-time, low-latency demands of edge platforms, inducing a perceptual saturation effect via severe observation redundancy.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.13245","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_jl1-cc-qa_e1b89c2a","familyId":"bmf_0a1d39bb805e","name":"JL1-CC&QA","oneLine":"JL1-CC&QA is a multi-task benchmark extending JL1-CD with change captioning and question answering. It includes 17,021 captions and 20,060 QA pairs over 5,000 bi-temporal image pairs from Jilin-1 satellite.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.31745","pdf":"https://arxiv.org/pdf/2606.31745","project":null,"code":"https://github.com/circleLZY/JL1-CD","data":null,"hfPaper":"https://huggingface.co/papers/2606.31745"},"evidence":{"snippet":"To bridge this semantic gap, we introduce JL1-CC&QA, a multi-task benchmark that extends the JL1-CD dataset with two complementary annotation layers: change captioning (CC) and change question answering (QA).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":129,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31745"},"ranking":{"90d":{"score":59,"rank":23,"coverage":0.55,"confidence":"Low"}},"description":"JL1-CC&QA is a multi-task benchmark extending JL1-CD with change captioning and question answering. It includes 17,021 captions and 20,060 QA pairs over 5,000 bi-temporal image pairs from Jilin-1 satellite.","whyItMatters":"Bridges the semantic gap in remote sensing change detection by providing captions and QA alongside binary masks. Enables multi-task understanding of surface changes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3bc6f9c58a01af83d374dcfeeddec50588500ff8840730c1af07bf7040025633"},"motivation":"Remote sensing change detection (CD) traditionally focuses on pixel-level binary segmentation, which identifies where changes occur but neither what nor why.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31745","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"circleLZY","organizationType":"academic-lab","sourceUrl":"https://github.com/circleLZY/JL1-CD","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_jmed48k_34101c20","familyId":"bmf_cfe906a6cd3c","name":"JMed48k","oneLine":"JMed48k evaluates vision-language models on 48,862 Japanese medical licensing exam questions from 11 national examinations (2005-2025), with images annotated under an 8-type taxonomy. The JMed48k-Eval subset contains 12,484 scored questions, including text-only and with-image items, scored separately.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22080","pdf":"https://arxiv.org/pdf/2605.22080","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.22080"},"evidence":{"snippet":"We introduce JMed48k, a multi-profession Japanese healthcare licensing benchmark for evaluating vision-language models.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.22080"},"ranking":{},"description":"JMed48k evaluates vision-language models on 48,862 Japanese medical licensing exam questions from 11 national examinations (2005-2025), with images annotated under an 8-type taxonomy. The JMed48k-Eval subset contains 12,484 scored questions, including text-only and with-image items, scored separately.","whyItMatters":"Provides a profession-stratified evaluation for vision-language models in Japanese medical licensing, enabling comparison of image use across professions and model types, and addressing the lack of multilingual medical benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c18c24c9f88a6eb18bbd546b52bcd69f7581db89456374faed93534ee5cbb963"},"motivation":"We introduce JMed48k, a multi-profession Japanese healthcare licensing benchmark for evaluating vision-language models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.22080","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"JMed48k Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2605.22080","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_dadf2d7a89695d09","familyId":"catalog_family_dadf2d7a89695d09","name":"JobBench","oneLine":"Job Bench evaluates AI agents on realistic professional tasks that require multi-step planning, research, and production of work artifacts.","description":"Job Bench evaluates AI agents on realistic professional tasks that require multi-step planning, research, and production of work artifacts.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Productivity","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2605.26329","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dadf2d7a89695d09"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/jobbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/job-bench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"jobBench","url":"https://benchlm.ai/benchmarks/jobbench","paperUrl":"https://arxiv.org/abs/2605.26329","year":"2026","fullName":"JobBench","format":"Agentic workplace deliverables","tasks":"130 tasks across 35 occupations","successorKey":null},{"catalog":"llm-stats","sourceId":"job-bench","url":"https://llm-stats.com/benchmarks/job-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","productivity","reasoning","agents"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_jor-bench_97431ead","familyId":"bmf_872a8906633f","name":"JOR-Bench","oneLine":"JOR-Bench is a collection of five Japanese-language benchmarks for LLMs in operations research, covering 1,319 problems from IndustryOR, MAMO, NL4OPT, OptiBench, and OptMATH, with pairs of Japanese problem statements and numerical answers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-18","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.16777","pdf":"https://arxiv.org/pdf/2607.16777","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.16777"},"evidence":{"snippet":"We present JOR-Bench, a collection of five Japanese-language benchmarks for evaluating the ability of large language models (LLMs) to formulate and solve operations research (OR) problems.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.16777"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"JOR-Bench is a collection of five Japanese-language benchmarks for LLMs in operations research, covering 1,319 problems from IndustryOR, MAMO, NL4OPT, OptiBench, and OptMATH, with pairs of Japanese problem statements and numerical answers.","whyItMatters":"Provides a standardized Japanese-language evaluation for OR formulation and solving, enabling cross-lingual comparison and highlighting language-specific issues.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bb4f2158b0200316b5162cf22af86bd22847fe5bd9bb282d9a99e947905e267e"},"motivation":"We present JOR-Bench, a collection of five Japanese-language benchmarks for evaluating the ability of large language models (LLMs) to formulate and solve operations research (OR) problems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.16777","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"The authors","organizationType":"community","sourceUrl":"https://arxiv.org/abs/2607.16777","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_judgmentbench_729b2b9c","familyId":"bmf_4b9658370bd4","name":"JudgmentBench","oneLine":"A dataset of 30 legal tasks with rubric scores and pairwise preference judgments from practicing attorneys, used to compare rubric-based scoring and comparative judgment.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-05-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.25240","pdf":"https://arxiv.org/pdf/2605.25240","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.25240"},"evidence":{"snippet":"We release JudgmentBench, a benchmark of 30 real-world legal tasks, paired with 1,539 rubric scores and 1,530 pairwise preference judgments collected from practicing attorneys--including at major U.S.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25240"},"ranking":{},"description":"A dataset of 30 legal tasks with rubric scores and pairwise preference judgments from practicing attorneys, used to compare rubric-based scoring and comparative judgment.","whyItMatters":"Supports research on expert judgment elicitation and aggregation in domains without ground truth.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6b8af1641e28f6c294d26b4cb93e1748d5c39742fa028ea514761628a1273283"},"motivation":"Two methodologies dominate current practices of benchmarking: rubric-based scoring evaluates items against predefined criteria, whereas comparative judgment elicits pairwise preferences between outputs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25240","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_k-browsecomp_c3715567","familyId":"bmf_7a5a02770094","name":"K-BrowseComp","oneLine":"K-BrowseComp evaluates web-browsing agents on 400 Korean-context problems, requiring multi-hop or parallel evidence retrieval from public Korean websites and returning a single short answer. Includes a 300-problem manually verified subset and a 100-problem synthetic stress-test split.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["Agents"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02404","pdf":"https://arxiv.org/pdf/2606.02404","project":null,"code":"https://github.com/prometheus-eval/K-BrowseComp","data":null,"hfPaper":"https://huggingface.co/papers/2606.02404"},"evidence":{"snippet":"We introduce K-BrowseComp, a web-browsing agent benchmark grounded in Korean contexts, consisting of 400 problems.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":59,"hfDailySubmittedAt":null,"githubStars":14,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02404"},"ranking":{},"description":"K-BrowseComp evaluates web-browsing agents on 400 Korean-context problems, requiring multi-hop or parallel evidence retrieval from public Korean websites and returning a single short answer. Includes a 300-problem manually verified subset and a 100-problem synthetic stress-test split.","whyItMatters":"Existing agentic benchmarks overlook Korean-language browsing, and frontier models show a significant performance drop on this benchmark. It provides a public protocol for measuring Korean web navigation and evidence-tracking abilities, which is relevant for deploying agents in Korean-language settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5809d0ceaf4b49ab9433161381ed8445813f7968c9c9ca3a4c74fd5299c03617"},"motivation":"Frontier model evaluations are shifting from foundational capabilities (e.g., instruction following and reasoning) toward compositional, agentic ones, but Korean agentic benchmarks remain scarce.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02404","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"prometheus-eval","organizationType":"community","sourceUrl":"https://github.com/prometheus-eval/K-BrowseComp","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_k-finhallu_490348d5","familyId":"bmf_44667dfd90e2","name":"K-FinHallu","oneLine":"Evaluates hallucination detection in multi-turn Korean financial RAG dialogues, with a taxonomy based on context answerability. Includes training and test splits for fine-tuning and benchmarking detectors.","area":"Vision & 3D","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29523","pdf":"https://arxiv.org/pdf/2605.29523","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29523"},"evidence":{"snippet":"We introduce K-FinHallu, the first benchmark for hallucination detection in multi-turn Korean financial RAG.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29523"},"ranking":{},"description":"Evaluates hallucination detection in multi-turn Korean financial RAG dialogues, with a taxonomy based on context answerability. Includes training and test splits for fine-tuning and benchmarking detectors.","whyItMatters":"Targets a high-stakes multilingual domain where existing benchmarks lack coverage. Provides a resource to improve hallucination detection and refusal behavior in financial applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"1f9402de42a9fc715cd5223d519e4a885bbd401e306e81d9e08423a66ff64b36"},"motivation":"Large Language Models (LLMs) have advanced financial automation through Retrieval-Augmented Generation (RAG), yet hallucinations remain a critical barrier to deployment in high-stakes environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29523","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"K-FinHallu Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2605.29523","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_k9-bench_cbdfe117","familyId":"bmf_b14e23a954dd","name":"K9-Bench","oneLine":"K9-Bench is a benchmark focused on real-world domestic dog videos, with approximately 5,000 question-answer pairs across 907 videos spanning 5 task categories. The tasks test long-form, canine-centric multimodal reasoning, including action and interaction understanding. A VLM/LLM-powered pipeline was used for data generation.","area":"Multimodal","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.02680","pdf":"https://arxiv.org/pdf/2607.02680","project":"https://ogmenrobotics.github.io/K9Bench","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.02680"},"evidence":{"snippet":"We introduce K9-Bench, a novel benchmark focused on real-world domestic dog videos, specifically targeting canine action and interaction understanding via approximately 5000 question-answer pairs across 907 videos spanning 5 distinct task categories that test long-form, canine-centric multimodal reasoning in MLLMs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02680"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"K9-Bench is a benchmark focused on real-world domestic dog videos, with approximately 5,000 question-answer pairs across 907 videos spanning 5 task categories. The tasks test long-form, canine-centric multimodal reasoning, including action and interaction understanding. A VLM/LLM-powered pipeline was used for data generation.","whyItMatters":"It addresses the underexplored application of multimodal models to animal-centric scenarios, providing a benchmark to evaluate models on recognizing distress signals and enabling responsive robotic companions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c4078098e6a609a824fa7ca61f8db7dfa08f76da36f9547bbbda889098c28934"},"motivation":"MLLMs have shown strong zero-shot capabilities across diverse inputs such as across images, video, audio, and text.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02680","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_aee8b5fdc1e6d419","familyId":"catalog_family_aee8b5fdc1e6d419","name":"Kernel Bench L3","oneLine":"Kernel Bench L3 evaluates agentic GPU kernel optimization across 50 problems. Qwen reports two metrics for this benchmark: median per-problem speedup over the PyTorch eager reference and the fraction of problems faster than torch.compile.","description":"Kernel Bench L3 evaluates agentic GPU kernel optimization across 50 problems. Qwen reports two metrics for this benchmark: median per-problem speedup over the PyTorch eager reference and the fraction of problems faster than torch.compile.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code","Systems"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/kernel-bench-l3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_aee8b5fdc1e6d419"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/kernel-bench-l3"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"kernel-bench-l3","url":"https://llm-stats.com/benchmarks/kernel-bench-l3","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code","systems"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_8e6e2dc857d11a64","familyId":"catalog_family_8e6e2dc857d11a64","name":"KernelBench","oneLine":"An agentic GPU-kernel benchmark that measures how much of the hardware roofline a model's correct, audit-clean kernels reach on six demanding CUDA and Triton problems.","description":"An agentic GPU-kernel benchmark that measures how much of the hardware roofline a model's correct, audit-clean kernels reach on six demanding CUDA and Triton problems.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://kernelbench.com/hard?gpu=h100","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8e6e2dc857d11a64"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/kernelbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"kernelBench","url":"https://benchlm.ai/benchmarks/kernelbench","paperUrl":"https://kernelbench.com/hard?gpu=h100","year":"2026","fullName":"KernelBench Hard H100","format":"Mean peak fraction of hardware roofline over valid cells","tasks":"6 GPU-kernel optimization problems","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_5fe3672fd4c09c4f","familyId":"catalog_family_5fe3672fd4c09c4f","name":"KernelBench Hard","oneLine":"KernelBench Hard evaluates agentic GPU kernel optimization on the hardest problem set. Each question is scored by the agent's submitted operator TFLOPs relative to the theoretical peak of the current hardware, with the benchmark score being the average across all questions.","description":"KernelBench Hard evaluates agentic GPU kernel optimization on the hardest problem set. Each question is scored by the agent's submitted operator TFLOPs relative to the theoretical peak of the current hardware, with the benchmark score being the average across all questions.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Agents","Code","Systems"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/MiniMaxAI/MiniMax-M3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5fe3672fd4c09c4f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/kernelbenchhard"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/kernelbench-hard"}],"catalogSources":[{"catalog":"benchlm","sourceId":"kernelBenchHard","url":"https://benchlm.ai/benchmarks/kernelbenchhard","paperUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","year":"2026","fullName":"KernelBench Hard","format":"Task success rate","tasks":"Hard GPU kernel coding tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"kernelbench-hard","url":"https://llm-stats.com/benchmarks/kernelbench-hard","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","agents","code","systems"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_723e107c4b61a832","familyId":"catalog_family_723e107c4b61a832","name":"KernelGen 1P","oneLine":"KernelGen 1P is an OpenAI AI-self-improvement evaluation that measures whether models can write and optimize compute kernels, part of the suite tracking progress toward accelerating internal research.","description":"KernelGen 1P is an OpenAI AI-self-improvement evaluation that measures whether models can write and optimize compute kernels, part of the suite tracking progress toward accelerating internal research.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code","Systems"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/kernelgen-1p","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_723e107c4b61a832"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/kernelgen-1p"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"kernelgen-1p","url":"https://llm-stats.com/benchmarks/kernelgen-1p","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code","systems"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_kernelgenbench_691527e5","familyId":"bmf_12af6c0291ed","name":"KernelGenBench","oneLine":"KernelGenBench evaluates LLM- and agent-generated Triton kernels across 210 operators from three sources (ATen, vLLM, cuBLAS) and six hardware platforms, with automatic accuracy verification and two evaluation tracks (LLM Pass@K and iterative agent generation).","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27231","pdf":"https://arxiv.org/pdf/2607.27231","project":null,"code":"https://github.com/flagos-ai/KernelGenBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.27231"},"evidence":{"snippet":"We present KernelGenBench, a unified benchmark for systematically evaluating LLM- and agent-generated Triton kernels across diverse operator sources and heterogeneous hardware platforms.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":14,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27231"},"ranking":{"90d":{"score":43,"rank":130,"coverage":0.55,"confidence":"Low"}},"description":"KernelGenBench evaluates LLM- and agent-generated Triton kernels across 210 operators from three sources (ATen, vLLM, cuBLAS) and six hardware platforms, with automatic accuracy verification and two evaluation tracks (LLM Pass@K and iterative agent generation).","whyItMatters":"Kernel generation is a specialized task lacking standardized evaluation; this benchmark provides a multi-source, multi-chip protocol to compare methods across diverse operators and hardware, enabling cost and portability assessment for autonomous kernel development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"05d9b8c03fdb9a325e3dd3710de2b9132f4d6a92b07747954a68759928335d5c"},"motivation":"Large language models (LLMs) have significantly increased the demand for efficient accelerator kernels, but kernel development remains a highly specialized and labor-intensive task.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27231","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FlagOS AI","organizationType":"company-research-lab","sourceUrl":"https://github.com/flagos-ai/KernelGenBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_keyframe-compass_c03623fb","familyId":"bmf_6d5f810b3b48","name":"KeyFrame-Compass","oneLine":"KeyFrame-Compass evaluates keyframe-conditioned video generation across 386 curated samples spanning three application domains, two video structures, two prompt granularities, two conditioning formats, and four keyframe densities. It jointly measures keyframe execution (presence, fidelity, temporal ordering, localization, persistence, uniqueness) and overall video quality via evidence-grounded MLLM judgments and specialized perception models.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14202","pdf":"https://arxiv.org/pdf/2607.14202","project":null,"code":"https://github.com/cactusqq/KeyFrame-Compass","data":null,"hfPaper":"https://huggingface.co/papers/2607.14202"},"evidence":{"snippet":"We present KeyFrame-Compass, the first comprehensive benchmark for evaluating keyframe-conditioned video generation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":43,"hfDailySubmittedAt":null,"githubStars":10,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14202"},"ranking":{"90d":{"score":43,"rank":128,"coverage":0.7,"confidence":"Medium"}},"description":"KeyFrame-Compass evaluates keyframe-conditioned video generation across 386 curated samples spanning three application domains, two video structures, two prompt granularities, two conditioning formats, and four keyframe densities. It jointly measures keyframe execution (presence, fidelity, temporal ordering, localization, persistence, uniqueness) and overall video quality via evidence-grounded MLLM judgments and specialized perception models.","whyItMatters":"As keyframe-based workflows grow in video production, there is no standardized way to assess whether models faithfully reproduce prescribed keyframes while maintaining natural video quality. KeyFrame-Compass provides a controlled testbed that reveals trade-offs and degradation patterns, helping practitioners choose models based on constraint density and input format.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"df26a592cc7930a241131cbc5981ea9921587bfcbb0510b6b389ea433c5d1994"},"motivation":"Video generation increasingly relies on keyframe-based workflows, where creators specify a sequence of reference images to guide generation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14202","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"KeyFrame-Compass Team","organizationType":"academic-lab","sourceUrl":"https://github.com/cactusqq/KeyFrame-Compass","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_kidbench_e196ca1f","familyId":"bmf_d5c09a3ec477","name":"KIDBench","oneLine":"KIDBench evaluates child-facing safety of LLMs for ages 7-11 using realistic queries and multi-turn child-actor simulations, scored by an LLM-as-a-Judge rubric.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.25510","pdf":"https://arxiv.org/pdf/2605.25510","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.25510"},"evidence":{"snippet":"We introduce KIDBench, a benchmark for evaluating child-facing LLM safety for ages 7-11 using a developmental-psychology-grounded LLM-as-a-Judge rubric.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25510"},"ranking":{},"description":"KIDBench evaluates child-facing safety of LLMs for ages 7-11 using realistic queries and multi-turn child-actor simulations, scored by an LLM-as-a-Judge rubric.","whyItMatters":"It addresses a gap in LLM safety evaluation by focusing on age-appropriate responses for children, with a novel rubric and multi-turn assessment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b9f62aa683b7a3015a20ce3331eeaac39dc91d4faa87f4b9ea92a774f67e38e7"},"motivation":"Children increasingly have access to Large Language Models (LLMs), which may expose them to responses that are developmentally inappropriate or require age-sensitive safety, guidance, and boundaries.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25510","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_909dbc3fd617af28","familyId":"catalog_family_909dbc3fd617af28","name":"Kimi Claw 24/7","oneLine":"A Moonshot AI internal long-horizon agent benchmark for persistent professional coworking tasks.","description":"A Moonshot AI internal long-horizon agent benchmark for persistent professional coworking tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_909dbc3fd617af28"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/kimiclaw247"}],"catalogSources":[{"catalog":"benchlm","sourceId":"kimiClaw247","url":"https://benchlm.ai/benchmarks/kimiclaw247","paperUrl":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","year":"2026","fullName":"Kimi Claw 24/7 Bench","format":"Average pass rate across repeated OpenClaw runs","tasks":"17 professional scenarios, 610 evaluation points","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_07f435211e9e9d57","familyId":"catalog_family_07f435211e9e9d57","name":"Kimi Claw 24/7 Bench","oneLine":"Kimi Claw 24/7 Bench is Moonshot AI's in-house benchmark for evaluating long-horizon agentic performance in persistent, multi-day coworking tasks. It spans 17 professional scenarios across 610 evaluation points, covering software engineering, ML research, recruiting, trading, and marketing tasks executed through the OpenClaw harness.","description":"Kimi Claw 24/7 Bench is Moonshot AI's in-house benchmark for evaluating long-horizon agentic performance in persistent, multi-day coworking tasks. It spans 17 professional scenarios across 610 evaluation points, covering software engineering, ML research, recruiting, trading, and marketing tasks executed through the OpenClaw harness.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/kimi-claw-24-7-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_07f435211e9e9d57"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/kimi-claw-24-7-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"kimi-claw-24-7-bench","url":"https://llm-stats.com/benchmarks/kimi-claw-24-7-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_a9518b9406a1a0bb","familyId":"catalog_family_a9518b9406a1a0bb","name":"Kimi Code Bench v2","oneLine":"Kimi Code Bench v2 is Moonshot AI's in-house benchmark for evaluating coding agents on realistic software engineering tasks across 10+ mainstream programming languages and a production tech stack spanning backend services, infrastructure, performance engineering, systems programming, security, frontend development, and ML/data engineering.","description":"Kimi Code Bench v2 is Moonshot AI's in-house benchmark for evaluating coding agents on realistic software engineering tasks across 10+ mainstream programming languages and a production tech stack spanning backend services, infrastructure, performance engineering, systems programming, security, frontend development, and ML/data engineering.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a9518b9406a1a0bb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/kimicodebenchv2"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/kimi-code-bench-v2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"kimiCodeBenchV2","url":"https://benchlm.ai/benchmarks/kimicodebenchv2","paperUrl":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","year":"2026","fullName":"Kimi Code Bench v2","format":"Coding-agent pass rate","tasks":"Realistic coding-agent tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"kimi-code-bench-v2","url":"https://llm-stats.com/benchmarks/kimi-code-bench-v2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","agents","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_bb9363753ede1201","familyId":"catalog_family_bb9363753ede1201","name":"KINA","oneLine":"KINA is a knowledge-intensive evaluation that measures a model's breadth and depth of factual knowledge across academic and professional domains.","description":"KINA is a knowledge-intensive evaluation that measures a model's breadth and depth of factual knowledge across academic and professional domains.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/kina","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bb9363753ede1201"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/kina"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"kina","url":"https://llm-stats.com/benchmarks/kina","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","general"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_522e16fb4492775c","familyId":"catalog_family_522e16fb4492775c","name":"KindBench","oneLine":"A behavioral benchmark that tests psychological safety across sixteen adversarial multi-turn conversations covering emotional safety, identity, sycophancy, and value integrity.","description":"A behavioral benchmark that tests psychological safety across sixteen adversarial multi-turn conversations covering emotional safety, identity, sycophancy, and value integrity.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kindbench.com/methodology","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_522e16fb4492775c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/kindbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"kindBench","url":"https://benchlm.ai/benchmarks/kindbench","paperUrl":"https://www.kindbench.com/methodology","year":"2026","fullName":"KindBench Psychological Safety Benchmark","format":"Judge-scored behavioral audit with human review","tasks":"16 multi-turn scenarios, 72 criteria","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_kinebench_86d2d206","familyId":"bmf_3667c8221a21","name":"KineBench","oneLine":"KineBench evaluates embodied world models via kinematic grounding, using 20 manipulation tasks in ManiSkill3 and metrics like spectral arc length and manipulability index.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.RO"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19876","pdf":"https://arxiv.org/pdf/2607.19876","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19876"},"evidence":{"snippet":"To reduce this ambiguity, we present KineBench, an IDM-free closed-loop benchmark for EWMs, built upon an explicit kinematic grounding pipeline.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19876"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"KineBench evaluates embodied world models via kinematic grounding, using 20 manipulation tasks in ManiSkill3 and metrics like spectral arc length and manipulability index.","whyItMatters":"Evaluating physical consistency of embodied world models is challenging. A benchmark without IDMs reduces attribution ambiguity, supporting reliable assessment of generated videos.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b1ff49f020c23985b3d7c0775dcd927d284ad1109fd9a1d7dcdb2b95736463b3"},"motivation":"Evaluating the physical consistency of embodied world models(EWMs) is a critical open challenge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19876","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_8c3809ca249fdf35","familyId":"catalog_family_8c3809ca249fdf35","name":"KMMLU","oneLine":"Evaluates Korean expert-level knowledge across 45 subjects. 20% of questions require Korean cultural context.","description":"Evaluates Korean expert-level knowledge across 45 subjects. 20% of questions require Korean cultural context.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Korean"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2402.11548","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8c3809ca249fdf35"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/kmmlu"}],"catalogSources":[{"catalog":"benchlm","sourceId":"kmmlu","url":"https://benchlm.ai/benchmarks/kmmlu","paperUrl":"https://arxiv.org/abs/2402.11548","year":"2024","fullName":"Korean Massive Multitask Language Understanding","format":"Multiple choice questions","tasks":"35,030 questions","successorKey":null}],"catalogCategories":["korean"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_4f65ee9a29d419cb","familyId":"catalog_family_4f65ee9a29d419cb","name":"KMMLU-Hard","oneLine":"A filtered hard subset of KMMLU containing ~5,000 questions that most models get wrong.","description":"A filtered hard subset of KMMLU containing ~5,000 questions that most models get wrong.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Korean"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://github.com/daekeun-ml/evaluate-llm-on-korean-dataset","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4f65ee9a29d419cb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/kmmluhard"}],"catalogSources":[{"catalog":"benchlm","sourceId":"kmmluHard","url":"https://benchlm.ai/benchmarks/kmmluhard","paperUrl":"https://github.com/daekeun-ml/evaluate-llm-on-korean-dataset","year":"2025","fullName":"KMMLU-Hard","format":"Multiple choice questions","tasks":"~5,000 questions","successorKey":null}],"catalogCategories":["korean"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_34190150a99b5938","familyId":"catalog_family_34190150a99b5938","name":"KMMLU-Pro","oneLine":"Korean National Professional Licensure exams evaluating professional-grade knowledge.","description":"Korean National Professional Licensure exams evaluating professional-grade knowledge.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Korean"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://benchlm.ai/benchmarks/kmmlupro","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_34190150a99b5938"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/kmmlupro"}],"catalogSources":[{"catalog":"benchlm","sourceId":"kmmluPro","url":"https://benchlm.ai/benchmarks/kmmlupro","paperUrl":null,"year":null,"fullName":"KMMLU-Pro","format":"Professional licensure exams","tasks":"~2,500 questions","successorKey":null}],"catalogCategories":["korean"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_9378f680a1505ee1","familyId":"catalog_family_9378f680a1505ee1","name":"KMMLU-Redux","oneLine":"Cleaned KMMLU from national technical qualification exams, with errors removed, decontaminated, and deduplicated.","description":"Cleaned KMMLU from national technical qualification exams, with errors removed, decontaminated, and deduplicated.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Korean"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://benchlm.ai/benchmarks/kmmluredux","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9378f680a1505ee1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/kmmluredux"}],"catalogSources":[{"catalog":"benchlm","sourceId":"kmmluRedux","url":"https://benchlm.ai/benchmarks/kmmluredux","paperUrl":null,"year":null,"fullName":"KMMLU-Redux","format":"Technical multiple choice","tasks":"~3,500 questions","successorKey":null}],"catalogCategories":["korean"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_07e6321345cc6853","familyId":"catalog_family_07e6321345cc6853","name":"KoBALT","oneLine":"Evaluates advanced Korean linguistic competence.","description":"Evaluates advanced Korean linguistic competence.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Korean"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://benchlm.ai/benchmarks/kobalt","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_07e6321345cc6853"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/kobalt"}],"catalogSources":[{"catalog":"benchlm","sourceId":"kobalt","url":"https://benchlm.ai/benchmarks/kobalt","paperUrl":null,"year":null,"fullName":"Korean Benchmark for Advanced Linguistic Tasks","format":"Advanced linguistics","tasks":"Linguistics questions","successorKey":null}],"catalogCategories":["korean"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e5cd7e8a76c3d1cd","familyId":"catalog_family_e5cd7e8a76c3d1cd","name":"Korean CSAT","oneLine":"The Korean SAT exam.","description":"The Korean SAT exam.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Korean"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://benchlm.ai/benchmarks/koreancsat","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e5cd7e8a76c3d1cd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/koreancsat"}],"catalogSources":[{"catalog":"benchlm","sourceId":"koreanCsat","url":"https://benchlm.ai/benchmarks/koreancsat","paperUrl":null,"year":null,"fullName":"College Scholastic Ability Test (수능)","format":"Standardized test","tasks":"Multi-subject exam","successorKey":null}],"catalogCategories":["korean"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_kotlin-benchmark_jetbrains-2026","familyId":"bmf_11a0a2a20c95","name":"Kotlin Benchmark","oneLine":"Evaluates AI coding agents on real-world Kotlin tasks from nine open-source repositories. The benchmark uses reproducible Docker environments, regression tests, and a scoring protocol that awards pass only when all expected test transitions are met.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation","Software engineering"],"topics":["Coding Agents","Kotlin"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://blog.jetbrains.com/kotlin/2026/07/introducing-the-kotlin-benchmark-evaluate-ai-coding-agents-on-real-world-kotlin-tasks/","pdf":null,"project":"https://kotlinlang.org/benchmark/","code":"https://github.com/Kotlin/kotlin-swe-bench","data":null,"hfPaper":null},"evidence":{"snippet":"The Kotlin Benchmark evaluates AI coding agents on real-world Kotlin tasks using reproducible execution and regression tests.","reasonCodes":["official project release","reviewed non-arXiv source"]},"dataStatus":"primary-source-reviewed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":50,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"official-project","id":"jetbrains-kotlin-benchmark-2026"},"ranking":{"90d":{"score":52,"rank":56,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates AI coding agents on real-world Kotlin tasks from nine open-source repositories. The benchmark uses reproducible Docker environments, regression tests, and a scoring protocol that awards pass only when all expected test transitions are met.","whyItMatters":"This benchmark addresses the lack of reference evaluation for Kotlin-specific coding agents, providing a reproducible, task-level framework for comparing agent performance on real-world Kotlin issues and tracking progress.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-22T18:51:01.308665Z","inputHash":"12dd9616dd867f1805bf6e8da925809f8216ad775f20f690a97b3c405f36d32d"},"motivation":"Measure whether general coding agents can solve realistic Kotlin repository issues under a reproducible SWE-bench-style protocol.","constructionDetail":"JetBrains adapted the SWE-bench methodology to resolved issues from real Kotlin repositories, packaged with Harbor-compatible containers and regression tests.","metrics":[{"name":"Tasks","value":"105","note":"initial release snapshot"},{"name":"Repositories","value":"8","note":"initial release snapshot"}],"curation":{"state":"source-reviewed","reviewedAt":"2026-08-19","sources":["https://blog.jetbrains.com/kotlin/2026/07/introducing-the-kotlin-benchmark-evaluate-ai-coding-agents-on-real-world-kotlin-tasks/","https://kotlinlang.org/benchmark/methodology/","https://github.com/Kotlin/kotlin-swe-bench"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://blog.jetbrains.com/kotlin/2026/07/introducing-the-kotlin-benchmark-evaluate-ai-coding-agents-on-real-world-kotlin-tasks/","source":"official-project","evidenceLevel":"official","verifiedAt":"2026-08-19T00:00:00Z"},"releaseDates":{"firstPublicAt":"2026-07-08","paperV1At":null},"publishers":[{"name":"JetBrains","organizationType":"company-research-lab","sourceUrl":"https://github.com/Kotlin/kotlin-swe-bench","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_ksafe-mm_8d1d1de0","familyId":"bmf_f2d785a9edd4","name":"KSAFE-MM","oneLine":"KSAFE-MM evaluates multimodal LLM safety in Korean contexts, with 12 models tested on general and culture-specific safety risks, including jailbreak-style textual queries paired with local visual cues.","area":"Safety & Trustworthiness","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["Multimodal","Safety"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.28013","pdf":"https://arxiv.org/pdf/2605.28013","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28013"},"evidence":{"snippet":"This paper introduces KSAFE-MM, a benchmark for Korean multimodal safety evaluation that covers both general safety risks and culture-specific vulnerabilities.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28013"},"ranking":{},"description":"KSAFE-MM evaluates multimodal LLM safety in Korean contexts, with 12 models tested on general and culture-specific safety risks, including jailbreak-style textual queries paired with local visual cues.","whyItMatters":"Existing safety benchmarks are English-centric and ignore local cultural risks; KSAFE-MM provides a general-to-local pipeline for culturally grounded safety evaluation, revealing trade-offs between safety and over-refusal.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"66ceda06edc2c977604ffe5ad1044258a296b89d6dfdb939d360444ec8a71f6a"},"motivation":"Multimodal Large Language Models (MLLMs) exacerbate safety risks by introducing vulnerabilities across multiple modalities, such as language and vision.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28013","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_kvdiagnosis-a-diagnostic-benchmark-for-kv-_86d87177","familyId":"bmf_00adefbae7c6","name":"KVDiagnosis","oneLine":"Diagnoses KV-cache compression failures in long-context language models using paired runs and cache, likelihood, attention, and decoding measurements.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.09412","pdf":"https://arxiv.org/pdf/2608.09412","project":null,"code":"https://github.com/ChosenQC/KVDiagnosis","data":null,"hfPaper":null},"evidence":{"snippet":"We present KVDiagnosis, a diagnostic dataset and benchmark with three contributions.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09412"},"ranking":{"30d":{"score":23,"rank":155,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":359,"coverage":0.55,"confidence":"Low"}},"description":"Diagnoses KV-cache compression failures in long-context language models using paired runs and cache, likelihood, attention, and decoding measurements.","whyItMatters":"Provides failure-focused diagnostics that reveal specific causes of compression errors, moving beyond aggregate task scores.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"558a3f8d973e6b212e25ac4e45fdd7f843ecb729dd8197c129b32a4952dadbe5"},"motivation":"KV-cache compression reduces long-context memory, but aggregate task scores reveal neither which correct executions fail nor why.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"Released with code and data, and defines a diagnostic protocol with fixed rules and metrics; intended for ongoing model comparison.","canonicalNameSource":"abstract","canonicalNameEvidence":"We present KVDiagnosis, a diagnostic dataset and benchmark"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09412","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":80,"confidence":"Medium","horizon":"7d","reason":"Targets a timely problem in long-context LLM efficiency and provides a comprehensive public repository with detailed diagnostics."},"evaluationMode":"score_submission","publishers":[{"name":"ChosenQC","organizationType":"community","sourceUrl":"https://github.com/ChosenQC/KVDiagnosis","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning","Long Context & Memory"],"domainScope":"general"},{"id":"bm_l3cube-indicquest-v2_c9db7d34","familyId":"bmf_da7eeb9063e2","name":"L3Cube-IndicQuest v2","oneLine":"L3Cube-IndicQuest v2 is a multilingual QA benchmark for evaluating India-specific factual knowledge of LLMs. It contains 3,471 curriculum-grounded English QA pairs translated into 19 Indic languages, totaling 69,420 pairs across 20 languages.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15535","pdf":"https://arxiv.org/pdf/2608.15535","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.15535"},"evidence":{"snippet":"We present L3Cube-IndicQuest v2, a large-scale gold-standard multilingual question-answering benchmark for evaluating the India-specific factual knowledge of Large Language Models (LLMs).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15535"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"L3Cube-IndicQuest v2 is a multilingual QA benchmark for evaluating India-specific factual knowledge of LLMs. It contains 3,471 curriculum-grounded English QA pairs translated into 19 Indic languages, totaling 69,420 pairs across 20 languages.","whyItMatters":"There is a need for benchmarks evaluating factual knowledge in Indic languages. This benchmark provides a large-scale gold-standard dataset with multiple evaluation protocols, showing consistent rankings across judges.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dbe6a5bed92d7a7ac9db02690ec25c0a8a5bf50fa15bf132f0fc67ad5809f265"},"motivation":"We present L3Cube-IndicQuest v2, a large-scale gold-standard multilingual question-answering benchmark for evaluating the India-specific factual knowledge of Large Language Models (LLMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15535","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0263696dad81dd27","familyId":"catalog_family_0263696dad81dd27","name":"LAB all-pass (Anthropic harness)","oneLine":"Strict task success requiring every expert-written legal-work rubric criterion to pass.","description":"Strict task success requiring every expert-written legal-work rubric criterion to pass.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0263696dad81dd27"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/legalagentbenchallpass"}],"catalogSources":[{"catalog":"benchlm","sourceId":"legalAgentBenchAllPass","url":"https://benchlm.ai/benchmarks/legalagentbenchallpass","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Legal Agent Benchmark all-pass rate — Anthropic harness","format":"All-criteria pass rate","tasks":"1,235 legal-agent tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_a773644fe6d08081","familyId":"catalog_family_a773644fe6d08081","name":"LAB all-pass (Harvey held-out)","oneLine":"Harvey AI's strict held-out task success rate requiring every rubric criterion to pass.","description":"Harvey AI's strict held-out task success rate requiring every rubric criterion to pass.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a773644fe6d08081"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/legalagentbenchheldoutallpass"}],"catalogSources":[{"catalog":"benchlm","sourceId":"legalAgentBenchHeldoutAllPass","url":"https://benchlm.ai/benchmarks/legalagentbenchheldoutallpass","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Legal Agent Benchmark all-pass rate — Harvey held-out set","format":"All-criteria pass rate","tasks":"Harvey-held-out legal-agent tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_da97a7163dae8af3","familyId":"catalog_family_da97a7163dae8af3","name":"LAB criterion-pass (Anthropic harness)","oneLine":"Mean fraction of expert-written rubric criteria passed across legal-agent tasks.","description":"Mean fraction of expert-written rubric criteria passed across legal-agent tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_da97a7163dae8af3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/legalagentbenchcriterionpass"}],"catalogSources":[{"catalog":"benchlm","sourceId":"legalAgentBenchCriterionPass","url":"https://benchlm.ai/benchmarks/legalagentbenchcriterionpass","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Legal Agent Benchmark mean criterion-pass rate — Anthropic harness","format":"Mean criterion-pass rate","tasks":"1,235 legal-agent tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_02c6c5c0001abd86","familyId":"catalog_family_02c6c5c0001abd86","name":"LAB criterion-pass (Harvey held-out)","oneLine":"Harvey AI's mean criterion-level score on its held-out legal-agent evaluation.","description":"Harvey AI's mean criterion-level score on its held-out legal-agent evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_02c6c5c0001abd86"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/legalagentbenchheldoutcriterionpass"}],"catalogSources":[{"catalog":"benchlm","sourceId":"legalAgentBenchHeldoutCriterionPass","url":"https://benchlm.ai/benchmarks/legalagentbenchheldoutcriterionpass","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Legal Agent Benchmark mean criterion-pass rate — Harvey held-out set","format":"Mean criterion-pass rate","tasks":"Harvey-held-out legal-agent tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_3f2ae1620bac7104","familyId":"catalog_family_3f2ae1620bac7104","name":"LABBench2","oneLine":"LABBench2 evaluates models on real-world biology research tasks.","description":"LABBench2 evaluates models on real-world biology research tasks.","area":"Agents & Tool Use","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Reasoning","Science","Agents","Biology"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/labbench2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3f2ae1620bac7104"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/labbench2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"labbench2","url":"https://llm-stats.com/benchmarks/labbench2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","science","agents","biology"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_labosbench_5eb589a2","familyId":"bmf_89ea60d8b827","name":"LabOSBench","oneLine":"LabOSBench evaluates multimodal GUI agents on 96 subtasks across eight web-based scientific-instrument simulators, covering workflows from sample loading to result inspection. Agents operate via a browser, with execution-based evaluation on task completion.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.16802","pdf":"https://arxiv.org/pdf/2606.16802","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.16802"},"evidence":{"snippet":"To this end, we introduce LabOSBench, a challenging benchmark for multimodal GUI agents built on a suite of web-based scientific-instrument simulators.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16802"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LabOSBench evaluates multimodal GUI agents on 96 subtasks across eight web-based scientific-instrument simulators, covering workflows from sample loading to result inspection. Agents operate via a browser, with execution-based evaluation on task completion.","whyItMatters":"Existing computer-use benchmarks focus on software tasks, leaving a gap for scientific instrument control. LabOSBench provides a safe, reproducible, low-cost testbed to assess agents' feedback-driven and long-horizon capabilities in instrument operation, supporting practical adoption in laboratory automation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ddc0246d344550a2396fe637254122572a12cb621d64e6d5f4c3f52ffdb7aac5"},"motivation":"Current computer-use benchmarks primarily focus on software operation tasks in virtualized systems, whereas scientific instrumentation scenarios require coordinated control over complex interfaces, and feedback-driven parameter adjustment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.16802","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_ladbench_44ed9c65","familyId":"bmf_3e35e374df54","name":"LADBench","oneLine":"LADBench evaluates large vision-language models on detecting logical anomalies in synthetic images across four domains: Residential, Urban, Collaborative, and Nature. It uses a Tiered Prompting Protocol with three progressive disclosure levels and automated scoring.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.17433","pdf":"https://arxiv.org/pdf/2606.17433","project":null,"code":null,"data":"https://huggingface.co/datasets/SahasraK/LADBench","hfPaper":"https://huggingface.co/papers/2606.17433"},"evidence":{"snippet":"To address this, we introduce LAD-bench, a benchmark of more than 1,000 curated synthetic images with logical anomalies across four domains: Residential, Urban, Collaborative, and Nature.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":76,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2606.17433"},"ranking":{"90d":{"score":46,"rank":100,"coverage":0.3,"confidence":"Low","datasetDownloadRank":49,"datasetRankPopulation":66}},"description":"LADBench evaluates large vision-language models on detecting logical anomalies in synthetic images across four domains: Residential, Urban, Collaborative, and Nature. It uses a Tiered Prompting Protocol with three progressive disclosure levels and automated scoring.","whyItMatters":"Existing anomaly benchmarks focus on visual errors, not the physical and social common sense required for open-world deployment. LADBench quantifies how much explicit assistance models need to localize and reason about logical faults, addressing a gap in evaluating sequential multimodal reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f25d55dd9bb8f8885b5281bc5e8cf2438f1c259ac5418c1a484844da40b2d5fe"},"motivation":"Large Vision Language Models (VLMs) excel at visual question answering and semantic grounding, but their capacity for autonomous logical reasoning remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"IEEE International Conference on Development and Learning (ICDL 2026)","evidence":"Accepted to the IEEE International Conference on Development and Learning (ICDL 2026)","evidenceUrl":"https://arxiv.org/abs/2606.17433","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"IEEE International Conference on Development and Learning (ICDL 2026)","reviewStatus":"accepted","decisionRaw":"Accepted to the IEEE International Conference on Development and Learning (ICDL 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.17433","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to the IEEE International Conference on Development and Learning (ICDL 2026)","level":"author-claim"}]}],"publishers":[{"name":"LADBench Team","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/datasets/SahasraK/LADBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_lakeqa_b77b7993","familyId":"bmf_dfc10241a16b","name":"LakeQA","oneLine":"LakeQA evaluates search-centric question answering over a 9.5 TB data lake of Wikipedia and government data. Tasks require multi-hop reasoning across heterogeneous structured and unstructured sources, with expert-annotated answers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.10460","pdf":"https://arxiv.org/pdf/2606.10460","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.10460"},"evidence":{"snippet":"To this end, we introduce LakeQA, a comprehensive benchmark for search-centric question answering over data lakes that jointly emphasizes searching and reasoning capabilities.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.10460"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LakeQA evaluates search-centric question answering over a 9.5 TB data lake of Wikipedia and government data. Tasks require multi-hop reasoning across heterogeneous structured and unstructured sources, with expert-annotated answers.","whyItMatters":"Existing QA benchmarks provide explicit evidence or trivial retrieval, missing the challenge of locating and composing evidence in large-scale data lakes. LakeQA fills this gap, supporting development and assessment of agents that can search and reason over massive heterogeneous data.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b3684db45f5e76dcb55cae665a8c38bf62640d81127db03fd9d426e64b9d8c86"},"motivation":"Recent large language models (LLMs) have shown rapid progress in reading-based question answering (QA), where evidence is explicitly provided or can be trivially retrieved.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.10460","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_lakequest_19d4b53f","familyId":"bmf_d4717a70a8cc","name":"LakeQuest","oneLine":"LakeQuest evaluates end-to-end question answering over data lakes with 9,846 QA pairs across three domains (AI/ML metadata, retail banking, biomedical drug info), with exact modality-aware evidence pointers, measuring retrieval and cross-modal synthesis.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences","Finance & Economics"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech","Financial Services"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.12310","pdf":"https://arxiv.org/pdf/2607.12310","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.12310"},"evidence":{"snippet":"To bridge this gap, we introduce LakeQuest, a human-validated benchmark of 9,846 QA pairs designed to evaluate the end-to-end retrieve-and-synthesize pipeline over realistic data lakes.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.12310"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LakeQuest evaluates end-to-end question answering over data lakes with 9,846 QA pairs across three domains (AI/ML metadata, retail banking, biomedical drug info), with exact modality-aware evidence pointers, measuring retrieval and cross-modal synthesis.","whyItMatters":"This benchmark fills the gap in evaluating QA systems on heterogeneous, weakly structured data lakes, exposing failure modes where high-quality retrieval does not guarantee correct reasoning, which is crucial for agentic QA development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fb583d1c3855871c5ebdd7aef9d3540e49cecbff97fa81d849b6ae3193b0be9e"},"motivation":"While modern question answering (QA) systems excel on clean, schema-aligned corpora, real-world knowledge is rarely so neatly packaged.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"Conference on Language Modeling (COLM) 2026","evidence":"24 pages, 4 figures, 18 tables. Accepted at the Conference on Language Modeling (COLM) 2026","evidenceUrl":"https://arxiv.org/abs/2607.12310","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"Conference on Language Modeling (COLM) 2026","reviewStatus":"accepted","decisionRaw":"24 pages, 4 figures, 18 tables. Accepted at the Conference on Language Modeling (COLM) 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.12310","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"24 pages, 4 figures, 18 tables. Accepted at the Conference on Language Modeling (COLM) 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"bm_langchoicebench_33a5b10b","familyId":"bmf_81dd6396539d","name":"LangChoiceBench","oneLine":"LangChoiceBench measures Python preference in project-level code generation across 28 projects and seven software areas, assessing language choice, recommendation-implementation consistency, and language diversity.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Code generation"],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.06041","pdf":"https://arxiv.org/pdf/2608.06041","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.06041"},"evidence":{"snippet":"To bridge this gap, we introduce LangChoiceBench, a project-level code-generation benchmark for measuring Python preference, recommendation-implementation consistency, and language diversity.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06041"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LangChoiceBench measures Python preference in project-level code generation across 28 projects and seven software areas, assessing language choice, recommendation-implementation consistency, and language diversity.","whyItMatters":"Addresses the lack of systematic evaluation of language preference in LLMs, providing a way to compare models on code generation beyond correctness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6ed3bea83cd13cb456ef17e28687d706519bcad6fee3d73d79d6082555c6a14c"},"motivation":"Large language models (LLMs) have been shown to exhibit strong Python preferences when generating project-level code, but there is currently no systematic way to measure this behaviour across new models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06041","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_layerrag-bench_a4840c44","familyId":"bmf_046f75fe617f","name":"LayerRAG-Bench","oneLine":"Evaluates cross-layer reliability of agentic RAG systems on 240 tasks across 8 enterprise domains, with 9 fault scenarios and 2 contract modes; measures success at evidence, tool-contract, authorization, and session-state layers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27353","pdf":"https://arxiv.org/pdf/2607.27353","project":null,"code":"https://github.com/MusaShams/layerrag-bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.27353"},"evidence":{"snippet":"We introduce LayerRAG-Bench, a controlled cross-layer reliability benchmark with 8 enterprise domains, 240 tasks, 9 fault scenarios, 2 contract modes, and 38,880 live task-level records across nine models from OpenAI, Anthropic, and Gemini.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27353"},"ranking":{"90d":{"score":28,"rank":271,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates cross-layer reliability of agentic RAG systems on 240 tasks across 8 enterprise domains, with 9 fault scenarios and 2 contract modes; measures success at evidence, tool-contract, authorization, and session-state layers.","whyItMatters":"Groundedness alone misses operational failures. This benchmark isolates which layer a mitigation repairs, supporting targeted reliability improvements rather than blanket fixes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9b981f5fecedc0de8814da2481491e65ecb24ccf2d66b04126e0e0d6aada9049"},"motivation":"Agentic retrieval-augmented generation systems can produce answers that appear grounded while failing at the evidence, tool-contract, authorization, or session-state layer.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27353","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Musa Shams","organizationType":"community","sourceUrl":"https://github.com/MusaShams/layerrag-bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_layoutbench_a9f59f74","familyId":"bmf_f115fe1b3e4f","name":"LayoutBench","oneLine":"LayoutBench evaluates three cloud storage layout strategies (individual objects, tar archives, Parquet columns) for multimedia data retrieval, measuring retrieval time, data transferred, and cost.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.DC"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.28880","pdf":"https://arxiv.org/pdf/2607.28880","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.28880"},"evidence":{"snippet":"We present LayoutBench, the first benchmark designed to fill this gap.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.28880"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LayoutBench evaluates three cloud storage layout strategies (individual objects, tar archives, Parquet columns) for multimedia data retrieval, measuring retrieval time, data transferred, and cost.","whyItMatters":"This benchmark fills a gap in storage benchmarking for multimedia workloads, but it is not a model evaluation benchmark.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"99c00b3ab4c9c4fa24a37e19392ec8e41f9cef5d8f60b322fd98783f671193e6"},"motivation":"Modern multimedia machine learning workloads increasingly store large-scale datasets in cloud object storage services such as AWS S3.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.28880","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_512407cb147ffc6d","familyId":"catalog_family_512407cb147ffc6d","name":"LBPP (v2)","oneLine":"LBPP (v2) benchmark - specific documentation not found in official sources, possibly related to language-based planning problems","description":"LBPP (v2) benchmark - specific documentation not found in official sources, possibly related to language-based planning problems","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/lbpp-(v2)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_512407cb147ffc6d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/lbpp-(v2)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"lbpp-(v2)","url":"https://llm-stats.com/benchmarks/lbpp-(v2)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_lcs-bench_45178465","familyId":"bmf_5c62eda733ab","name":"LCS-Bench","oneLine":"LCS-Bench is a theory-scale benchmark for auto-formalization in logics for computer science. It includes 327 textbook items, over 4,076 Lean declarations, and supports five evaluation tracks with definitional equivalence checkers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26525","pdf":"https://arxiv.org/pdf/2606.26525","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.26525"},"evidence":{"snippet":"In this paper, we introduce LCS-Bench, a stand-alone, theory-scale benchmark based on Logics for Computer Science.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26525"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LCS-Bench is a theory-scale benchmark for auto-formalization in logics for computer science. It includes 327 textbook items, over 4,076 Lean declarations, and supports five evaluation tracks with definitional equivalence checkers.","whyItMatters":"Auto-formalization at theory scale remains challenging. LCS-Bench provides a reusable benchmark to measure consistency, faithfulness, and correctness of formalization systems, enabling progress in scalable verification.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"357a38782c0ca8707c63929c38864bbe3eafa0f5b15f6979e30fc111927b4d94"},"motivation":"Auto-formalization is critical for scalable formal verification, but existing progress largely focuses on isolated statements, while theory-scale auto-formalization, which coherently translates hundreds of interdependent definitions, lemmas, and theorems, remains open due to challenges in consistency, faithfulness, scalability, and correctness.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.26525","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_ldu-bench_c9c6a21a","familyId":"bmf_3b6629c4eda5","name":"LDU-Bench","oneLine":"LDU-Bench evaluates multimodal LLMs on lithography defect understanding through four tasks: defect triage, morphology recognition, coarse localization, and image-conditioned cause analysis, using task-level metrics and the Lithography Closure Score.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03078","pdf":"https://arxiv.org/pdf/2608.03078","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03078"},"evidence":{"snippet":"To this end, this paper proposes LDU-Bench, a multi-task multimodal benchmark for lithography defect understanding.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03078"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LDU-Bench evaluates multimodal LLMs on lithography defect understanding through four tasks: defect triage, morphology recognition, coarse localization, and image-conditioned cause analysis, using task-level metrics and the Lithography Closure Score.","whyItMatters":"Existing industrial anomaly detection benchmarks focus on defect presence, but lithography review requires deeper understanding of morphology, location, and causes. This benchmark aims to quantify those capabilities in a unified platform.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"74676a0275c2c257e1187afb95402961fe5bdd89836397d35e084d00d13f8fd7"},"motivation":"Multimodal large language models have demonstrated strong defect recognition capability in industrial anomaly detection.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03078","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_59ee8d88e3ee6f02","familyId":"catalog_family_59ee8d88e3ee6f02","name":"Legal Agent Benchmark","oneLine":"The Legal Agent Benchmark (LAB) is Harvey's open-source benchmark for evaluating AI agents on complex, long-horizon legal work. Tasks are scored under an all-pass standard against expert-curated rubrics, where a task passes only if every required rubric criterion (facts, conclusions, citations, structure, and analytical moves) passes.","description":"The Legal Agent Benchmark (LAB) is Harvey's open-source benchmark for evaluating AI agents on complex, long-horizon legal work. Tasks are scored under an all-pass standard against expert-curated rubrics, where a task passes only if every required rubric criterion (facts, conclusions, citations, structure, and analytical moves) passes.","area":"Agents & Tool Use","applicationDomains":["Law & Government"],"primaryDomain":"Law & Government","industrySectors":[],"capabilities":[],"topics":["Legal","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/legal-agent-benchmark","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_59ee8d88e3ee6f02"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/legal-agent-benchmark"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"legal-agent-benchmark","url":"https://llm-stats.com/benchmarks/legal-agent-benchmark","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["legal","reasoning","agents"],"catalogModelCount":13,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_1b0e7637e7e50cfb","familyId":"catalog_family_1b0e7637e7e50cfb","name":"Legal Research Bench","oneLine":"Evaluating agents on legal research tasks across diverse areas of US law","description":"Evaluating agents on legal research tasks across diverse areas of US law","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/legal_research","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1b0e7637e7e50cfb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/legalresearchbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"legalResearchBench","url":"https://benchlm.ai/benchmarks/legalresearchbench","paperUrl":"https://www.vals.ai/benchmarks/legal_research","year":"2026","fullName":"Vals Legal Research Bench","format":"Accuracy score","tasks":"US-law legal research tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_8bda6ad33efc111f","familyId":"catalog_family_8bda6ad33efc111f","name":"LegalBench","oneLine":"Vals AI legal benchmark with issue, rule, conclusion, interpretation, and rhetoric task views.","description":"Vals AI legal benchmark with issue, rule, conclusion, interpretation, and rhetoric task views.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/legal_bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8bda6ad33efc111f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valslegalbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsLegalBench","url":"https://benchlm.ai/benchmarks/valslegalbench","paperUrl":"https://www.vals.ai/benchmarks/legal_bench","year":"2026","fullName":"Vals LegalBench","format":"Accuracy score","tasks":"Legal reasoning task views","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_legalcitetrust_1777e861","familyId":"bmf_9504ffa72e56","name":"LegalCiteTrust","oneLine":"LegalCiteTrust evaluates citation trustworthiness in Chinese long-form legal research reports, assessing coverage, support, and citation-level existence, fidelity, and applicability.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20872","pdf":"https://arxiv.org/pdf/2607.20872","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20872"},"evidence":{"snippet":"We introduce LegalCiteTrust, a benchmark for evaluating citation trustworthiness in Chinese long-form legal research reports.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20872"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LegalCiteTrust evaluates citation trustworthiness in Chinese long-form legal research reports, assessing coverage, support, and citation-level existence, fidelity, and applicability.","whyItMatters":"Long-form legal research reports increasingly rely on LLMs, but citation trustworthiness is critical for legal accuracy. This benchmark addresses the gap by measuring whether citations are not only real but also accurate and applicable, providing a more nuanced evaluation than simple existence checks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dc710f6b8c90ca698f074b46bf34a378659c696fc4d2ad2f6f710e0cae2e1a56"},"motivation":"Long-form legal research reports increasingly rely on LLMs and agentic research systems, but their reliability depends not only on answering the task, but also on whether cited legal authorities are trustworthy.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20872","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_lethe_74a00c13","familyId":"bmf_7b3dcb32b39a","name":"Lethe","oneLine":"Lethe is a benchmark for federated unlearning in medical imaging. It evaluates twelve methods across eight task families, including classification, segmentation, denoising, cross-modality synthesis, and vision-language question answering. Three forgetting granularities (hospital, class, patient) are assessed against a retrained gold standard on utility, privacy, and cost.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.01094","pdf":"https://arxiv.org/pdf/2608.01094","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01094"},"evidence":{"snippet":"We present Lethe, a benchmark for federated unlearning in medical imaging.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01094"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Lethe is a benchmark for federated unlearning in medical imaging. It evaluates twelve methods across eight task families, including classification, segmentation, denoising, cross-modality synthesis, and vision-language question answering. Three forgetting granularities (hospital, class, patient) are assessed against a retrained gold standard on utility, privacy, and cost.","whyItMatters":"Existing unlearning evaluations primarily use natural images, leaving unclear whether methods transfer to clinical data. Lethe provides a shared protocol for medical imaging, enabling comparison of unlearning methods across diverse tasks and forgetting difficulties.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ba26a31b8f3950da5befde45ab5c250da19fa34eeafe20e7478f2d9ec8d7a70a"},"motivation":"Federated learning enables medical-imaging models to be trained across hospitals, and privacy law, most explicitly the GDPR ``right to be forgotten'', turns removing a hospital's, a class's, or a patient's influence from such a model into a federated unlearning problem.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.01094","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_levante-bench_b9930c93","familyId":"bmf_4b5b1a7a3878","name":"LEVANTE-bench","oneLine":"LEVANTE-bench evaluates vision-language models on six cognitive tasks from the LEVANTE dataset, comparing model performance and error patterns with children aged 5-12 across three countries.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05497","pdf":"https://arxiv.org/pdf/2606.05497","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05497"},"evidence":{"snippet":"We present LEVANTE-bench, a benchmark based on tasks and data from the Learning Variability Network (LEVANTE), which distributes open-source tasks and data measuring children's cognition across languages and cultures.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05497"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LEVANTE-bench evaluates vision-language models on six cognitive tasks from the LEVANTE dataset, comparing model performance and error patterns with children aged 5-12 across three countries.","whyItMatters":"LEVANTE-bench addresses the gap in evaluating VLMs against human cognitive development, providing a structured comparison that can inform model design and understanding of alignment with human cognition.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"488d9807d6565ff19365194f5250dd2a46907f4353fb98eb5190f3f7e78c03a3"},"motivation":"Given the inherently multimodal nature of human experience, vision-language models (VLMs) hold substantial promise for modeling human cognition as it grows and develops with experience.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05497","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LEVANTE-bench Authors","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.05497","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_lexkairos_261e303a","familyId":"bmf_453e64d655a7","name":"LexKairos","oneLine":"LexKairos is a benchmark for evaluating the temporal capabilities of LLMs in the Chinese legal context, covering statutory temporal knowledge, case temporal modeling, and statute-case temporal reasoning across nine sub-tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Factuality"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09106","pdf":"https://arxiv.org/pdf/2608.09106","project":null,"code":"https://github.com/thunlp/LexKairos","data":null,"hfPaper":"https://huggingface.co/papers/2608.09106"},"evidence":{"snippet":"To address this gap, we propose LexKairos, a comprehensive benchmark for evaluating the temporal capabilities of LLMs in the Chinese legal context across three dimensions: statutory temporal knowledge, case temporal modeling, and statute-case temporal reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09106"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LexKairos is a benchmark for evaluating the temporal capabilities of LLMs in the Chinese legal context, covering statutory temporal knowledge, case temporal modeling, and statute-case temporal reasoning across nine sub-tasks.","whyItMatters":"Legal temporal capabilities are underexplored in existing legal AI benchmarks; LexKairos provides a public tool for evaluating time-sensitive legal reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d2eba5efeddac2ec01fea4127da96229a994117c9f0c30617482e507d044eda4"},"motivation":"Large language models (LLMs) have demonstrated strong performance across a wide range of legal tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09106","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"THUNLP","organizationType":"academic-lab","sourceUrl":"https://github.com/thunlp/LexKairos","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_lexrubric_7a24242e","familyId":"bmf_8419445f9469","name":"LexRubric","oneLine":"LexRubric evaluates open-ended legal tasks in Chinese, with 649 instances from legal consultation and judicial examination. It includes 12,337 expert-written atomic scoring criteria under a six-dimensional framework, enabling fine-grained diagnostic evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09389","pdf":"https://arxiv.org/pdf/2606.09389","project":null,"code":"https://github.com/foggpoy/LexRubric","data":null,"hfPaper":"https://huggingface.co/papers/2606.09389"},"evidence":{"snippet":"We introduce LexRubric, a rubric-based benchmark for evaluating open-ended Chinese legal tasks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09389"},"ranking":{"90d":{"score":20,"rank":386,"coverage":0.7,"confidence":"Medium"}},"description":"LexRubric evaluates open-ended legal tasks in Chinese, with 649 instances from legal consultation and judicial examination. It includes 12,337 expert-written atomic scoring criteria under a six-dimensional framework, enabling fine-grained diagnostic evaluation.","whyItMatters":"Open-ended legal responses require fine-grained evaluation beyond exact matching. LexRubric provides rubric-based diagnostic assessment, showing distinct capability profiles across models and highlighting challenges in open-ended legal questions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"56b0ecb68d99a353289d7c154ada1d195d40459f80b74ca556b558b82401028e"},"motivation":"As large language models (LLMs) are increasingly applied to real-world legal tasks, evaluating the reliability of their open-ended legal responses has become essential.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09389","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LexRubric Team","organizationType":"academic-lab","sourceUrl":"https://github.com/foggpoy/LexRubric","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_lh-avln_cb4be7f0","familyId":"bmf_bd018386fef4","name":"LH-AVLN","oneLine":"LH-AVLN is a benchmark for long-horizon navigation that combines multi-goal missions, heterogeneous goal specifications, and persistent spatialized audio cues. Agents must execute missions of two to four goals specified by category, language, or reference image, using RGB-D, pose, and binaural audio in indoor 3D environments. It supports ordered and unordered missions with alternating goal-associated sounds.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.03920","pdf":"https://arxiv.org/pdf/2607.03920","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.03920"},"evidence":{"snippet":"We introduce LH-AVLN, a benchmark for Long-Horizon Audio-Visual-Language Navigation that combines multi-goal mission execution, heterogeneous goal specifications, and persistent spatialized acoustic cues.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.03920"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LH-AVLN is a benchmark for long-horizon navigation that combines multi-goal missions, heterogeneous goal specifications, and persistent spatialized audio cues. Agents must execute missions of two to four goals specified by category, language, or reference image, using RGB-D, pose, and binaural audio in indoor 3D environments. It supports ordered and unordered missions with alternating goal-associated sounds.","whyItMatters":"Existing navigation benchmarks do not integrate long-horizon missions with audio-visual cues and heterogeneous goal types, while audio-visual tasks are typically single-goal. LH-AVLN addresses this gap by providing a multi-goal environment with acoustic guidance and distractors, enabling more realistic evaluation of embodied agents in complex missions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"980a8dd752a8b2cef97ca30f6aa3f5b1eeb18796d059e953b0118af450b16ed1"},"motivation":"Embodied navigation is moving toward long-horizon missions, yet existing long-horizon benchmarks are largely acoustically silent, and audio-visual navigation tasks typically focus on a single goal.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.03920","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_libad_b1abf215","familyId":"bmf_dac7b1dda460","name":"LIBAD","oneLine":"LIBAD is a multimodal anomaly detection dataset from Li-ion battery electrode manufacturing, providing aligned visible-light and X-ray radiography images. It includes benchmarks of methods under inline-compatible settings.","area":"Multimodal","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Manufacturing"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07958","pdf":"https://arxiv.org/pdf/2608.07958","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07958"},"evidence":{"snippet":"We introduce LIBAD, the first multimodal anomaly detection benchmark for Li-ion battery electrode manufacturing.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07958"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"LIBAD is a multimodal anomaly detection dataset from Li-ion battery electrode manufacturing, providing aligned visible-light and X-ray radiography images. It includes benchmarks of methods under inline-compatible settings.","whyItMatters":"If available, it could enable evaluation of anomaly detection in continuous manufacturing with weakly correlated modalities, but without a release path, its standalone value is unverified.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8851abd4cae76729fd2f845d26f84411a7d1dc9d2c85a70abbf0557ed4563887"},"motivation":"Multimodal industrial anomaly detection has largely focused on discrete products using strongly correlated RGB and 3D observations, leaving continuous process manufacturing and weakly correlated sensing modalities underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07958","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_libero-vifo_76763b3b","familyId":"bmf_092e28009a80","name":"LIBERO-VIFO","oneLine":"Evaluates whether vision-language-action models follow authorized visual cues while resisting unauthorized visual influence.","area":"Safety & Trustworthiness","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Embodied manipulation","Visual instruction following","Safety"],"topics":["Multimodal","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.17600","pdf":"https://arxiv.org/pdf/2608.17600","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.17600"},"evidence":{"snippet":"To address these gaps, we introduce LIBERO-VIFO, a benchmark to evaluate both the capability and safety of visual cue following in VLA models.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17600"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LIBERO-VIFO is a benchmark for visual cue following in vision-language-action models, evaluating capability and safety across eight cue families. However, no public artifacts or official links are provided.","whyItMatters":"If released, it could inform the development of safer VLA systems, but current lack of access limits its practical value.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"299804e0fc9f5979754ed09eb8b0c5057dff36aab38ecfdc75b78d811ac14934"},"motivation":"Visual cues are increasingly adopted to guide robot learning, but whether Vision-Language-Action (VLA) models can reliably follow authorized cues while disregarding unauthorized ones remains unclear.","constructionDetail":"LIBERO-VIFO tests authorized and unauthorized visual influence across robot-manipulation tasks and eight families of visual cues.","detail":{"taskBreakdown":["Text overlays","Geometric symbols","Ghost trajectories","Pictorial demonstrations","Hand gestures","Visual signage","Visual highlights","Machine-readable codes"],"protocol":{"tasks":"1,347 instances from 40 LIBERO tasks and 33 cue variants","primaryMetric":"Full-Chain Accuracy and authorized/unauthorized visual following rates","version":"arXiv v1"}},"curation":{"state":"source-reviewed","reviewedAt":"2026-08-20","sources":["https://arxiv.org/abs/2608.17600","https://arxiv.org/html/2608.17600"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17600","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"score_submission","availability":{"paperStatus":"available","githubStatus":"not_found","hfDatasetStatus":"not_found","evaluatorStatus":"not_found","submissionStatus":"not_found"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_libevobench_4c11745f","familyId":"bmf_0e8e47aebe59","name":"LibEvoBench","oneLine":"LibEvoBench is a multi-task benchmark for evaluating code generation models on API evolution across versions of Python libraries. It includes tasks spanning multiple library versions and introduces the Software Evolution Understanding Score (SEUS) to measure version-specific knowledge consistency.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation","Factuality"],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.25402","pdf":"https://arxiv.org/pdf/2606.25402","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.25402"},"evidence":{"snippet":"To systematically evaluate this phenomenon, we introduce LibEvoBench, a multi-task benchmark spanning multiple versions of widely used Python libraries, along with a new metric, the Software Evolution Understanding Score (SEUS), to measure models' consistency when working with evolving APIs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.25402"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LibEvoBench is a multi-task benchmark for evaluating code generation models on API evolution across versions of Python libraries. It includes tasks spanning multiple library versions and introduces the Software Evolution Understanding Score (SEUS) to measure version-specific knowledge consistency.","whyItMatters":"Addresses the gap in evaluating models on version-specific API knowledge, which is critical for real-world software maintenance. It highlights limitations of current training on temporally mixed corpora and motivates temporally grounded learning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7ae1075acff0c0c02977d3297cabbb264cd8d1fa8a08ba414458daa3221d339d"},"motivation":"Large software projects often depend on older versions of libraries, even as APIs continue to evolve across releases.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"DL4Code workshop at ICML 2026","evidence":"Accepted at the DL4Code workshop at ICML 2026","evidenceUrl":"https://arxiv.org/abs/2606.25402","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"DL4Code workshop at ICML 2026","reviewStatus":"accepted","decisionRaw":"Accepted at the DL4Code workshop at ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.25402","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at the DL4Code workshop at ICML 2026","level":"author-claim"}]}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_lifeplanner_df604ac8","familyId":"bmf_b50971360bb5","name":"LifePlanner","oneLine":"Evaluates LLM agents on geo-spatial planning tasks using map data enriched with social media posts, measuring pass rate across four task categories and three difficulty levels.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-28","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25039","pdf":"https://arxiv.org/pdf/2608.25039","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.25039"},"evidence":{"snippet":"We introduce LifePlanner, a benchmark that enriches map data with large-scale local social media posts and provides access through an MCP toolset.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25039"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLM agents on geo-spatial planning tasks using map data enriched with social media posts, measuring pass rate across four task categories and three difficulty levels.","whyItMatters":"Introduces realistic open-ended social signals into geo-spatial agent evaluation, revealing degradation in grounded planning beyond simple retrieval.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"fd887f0bde1922cc2835721415bce440a07dc3937fb21373c6519d22867981cc"},"motivation":"Geo-spatial planning, like trip design, is a realistic testbed for LLM agents because it requires grounded tool use, noisy evidence retrieval, and multi-constraint reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25039","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"attentionForecast":{"score":40,"confidence":"Low","horizon":"7d","reason":"Agentic planning benchmarks are topical, but no public data or code link is provided, limiting immediate reproducibility and early adoption."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d3d27b0f6f6ef01a","familyId":"catalog_family_d3d27b0f6f6ef01a","name":"LifeSciBench","oneLine":"LifeSciBench is an expert-authored, expert-reviewed benchmark of 750 open-ended life-science research tasks spanning seven workflows and seven biological domains. Responses are graded against 19,020 physician- and scientist-written rubric criteria rather than multiple-choice answers, and most tasks require interpreting attached artifacts such as figures, PDFs, and sequence files.","description":"LifeSciBench is an expert-authored, expert-reviewed benchmark of 750 open-ended life-science research tasks spanning seven workflows and seven biological domains. Responses are graded against 19,020 physician- and scientist-written rubric criteria rather than multiple-choice answers, and most tasks require interpreting attached artifacts such as figures, PDFs, and sequence files.","area":"Agents & Tool Use","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Science","Healthcare","Agents","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/lifescibench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d3d27b0f6f6ef01a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/lifescibench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"lifescibench","url":"https://llm-stats.com/benchmarks/lifescibench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","science","healthcare","agents","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_ligbench_47189b6a","familyId":"bmf_4c9a86e62e01","name":"LigBench","oneLine":"LigBench is an automated evaluation benchmark for AI research idea generation, using fine-grained and reliable evaluation across generation distributions. It includes PAIR-IQ, a dataset for training pairwise idea judgment models.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.13136","pdf":"https://arxiv.org/pdf/2608.13136","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.13136"},"evidence":{"snippet":"To address this challenge, we propose LigBench, an automated evaluation benchmark that enables fine-grained and reliable evaluation of AI research ideas, consistently applicable across different generation distributions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13136"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LigBench is an automated evaluation benchmark for AI research idea generation, using fine-grained and reliable evaluation across generation distributions. It includes PAIR-IQ, a dataset for training pairwise idea judgment models.","whyItMatters":"Current evaluation of research idea generation is fragmented and lacks objective standards. LigBench aims to provide stable, interpretable, and expert-aligned assessments to support scalable and objective evaluation in this emerging area.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e331c59c98129cd0071b460dfb9bb4845dc59ec2eceaedf25f39863c4730f9bf"},"motivation":"With the rapid advancement of large language models (LLMs), research idea generation has attracted increasing attention.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13136","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_lilybench_6840fe07","familyId":"bmf_91b640010f26","name":"LilyBench","oneLine":"LilyBench evaluates symbolic music generation and understanding using LilyPond, with a 200-prompt generation suite and ten understanding tasks, scored via compile rate, descriptor similarity, and FMD.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SD"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08722","pdf":"https://arxiv.org/pdf/2606.08722","project":null,"code":"https://github.com/CSCPadova/lilybench","data":null,"hfPaper":"https://huggingface.co/papers/2606.08722"},"evidence":{"snippet":"We introduce LilyBench, a LilyPond-based benchmark that jointly evaluates symbolic music generation and music understanding on the same family of open-weight LLMs.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08722"},"ranking":{"90d":{"score":15,"rank":406,"coverage":0.7,"confidence":"Medium"}},"description":"LilyBench evaluates symbolic music generation and understanding using LilyPond, with a 200-prompt generation suite and ten understanding tasks, scored via compile rate, descriptor similarity, and FMD.","whyItMatters":"Symbolic music evaluation is fragmented; LilyBench provides a joint benchmark with public datasets and code, enabling reproducible comparisons and metric triangulation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"75b31e09fe9c1b5c0864b81279efdeb6db326d7cca4fb175816fdbab6d07f515"},"motivation":"Symbolic music evaluation for large language models remains fragmented across representations, datasets, and metrics.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"Ital-IA 2026","evidence":"Accepted at Ital-IA 2026","evidenceUrl":"https://arxiv.org/abs/2606.08722","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"Ital-IA 2026","reviewStatus":"accepted","decisionRaw":"Accepted at Ital-IA 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.08722","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at Ital-IA 2026","level":"author-claim"}]}],"publishers":[{"name":"CSCPadova","organizationType":"academic-lab","sourceUrl":"https://github.com/CSCPadova/lilybench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_lingjing_2ca7c6e7","familyId":"bmf_da300d8a728c","name":"Lingjing","oneLine":"Lingjing is a simulation testbed for heterogeneous multi-agent embodied tasks in urban environments, with a Gym-like interface and support for natural-language missions. It includes engine-based evaluations and attribution-ready replays.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-08","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.08045","pdf":"https://arxiv.org/pdf/2608.08045","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.08045"},"evidence":{"snippet":"Lingjing provides a unified testbed that enables reproducible end-to-end evaluation and systematic failure diagnosis in urban multi-agent embodied intelligence.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08045"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Lingjing is a simulation testbed for heterogeneous multi-agent embodied tasks in urban environments, with a Gym-like interface and support for natural-language missions. It includes engine-based evaluations and attribution-ready replays.","whyItMatters":"If published, it could support reproducible evaluation of multi-agent urban embodied intelligence, but without a public release path, its standalone value is unverified.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"569f247cfcb5899a04416f64e8dba1b1935bd5286911092874fadbbdc287430b"},"motivation":"Urban embodied intelligence requires coordination among heterogeneous agents (e.g., UAVs, ground robots, and autonomous vehicles) in dynamic cities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.08045","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_d6d7a20390c04d5a","familyId":"catalog_family_d6d7a20390c04d5a","name":"LingoQA","oneLine":"A benchmark for multimodal spatial-language understanding and visual-linguistic question answering.","description":"A benchmark for multimodal spatial-language understanding and visual-linguistic question answering.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/lingoqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d6d7a20390c04d5a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/lingoqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"lingoqa","url":"https://llm-stats.com/benchmarks/lingoqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","multimodal","reasoning","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_on-the-limitations-of-cross-lingual-consis_d9fe8562","familyId":"bmf_b2859ed03288","name":"LingT2I","oneLine":"Evaluates cross-lingual text-to-image generation across content generation and text rendering tasks for 10 languages and 33K prompts.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.11002","pdf":"https://arxiv.org/pdf/2608.11002","project":null,"code":"https://github.com/RISys-Lab/LingT2I","data":null,"hfPaper":null},"evidence":{"snippet":"To fill this gap, we introduce LingT2I, a benchmark covering 10 widely used languages with 33K prompts, designed to evaluate cross-lingual effects in both content generation and text rendering.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11002"},"ranking":{"30d":{"score":19,"rank":163,"coverage":0.85,"confidence":"High"},"90d":{"score":25,"rank":296,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates cross-lingual text-to-image generation across content generation and text rendering tasks for 10 languages and 33K prompts.","whyItMatters":"Reveals cross-lingual inconsistencies in text-to-image models, providing a basis for studying linguistic inequality and improving multilingual generation.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"3a94879d8653de0d6681dfcbfac40d1ae26c3d2f1becdf545f250fe23319abd4"},"motivation":"Text-to-image (T2I) generation has achieved remarkable progress in recent years.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is explicitly named, has a public GitHub repository and Hugging Face dataset, and provides evaluation scripts for reproducible scoring.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce LingT2I, a benchmark covering 10 widely used languages with 33K prompts"},"publication":{"status":"acceptance_claimed","venue":"ACM MM 2026","evidence":"Accepted to ACM MM 2026","evidenceUrl":"https://arxiv.org/abs/2608.11002","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-19T11:05:13.395391Z"},"venueAttempts":[{"venueName":"ACM MM 2026","reviewStatus":"accepted","decisionRaw":"Accepted to ACM MM 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.11002","observedAt":"2026-08-19T11:05:13.395391Z","rawValue":"Accepted to ACM MM 2026","level":"author-claim"}]}],"attentionForecast":{"score":62,"confidence":"Medium","horizon":"7d","reason":"Multilingual text-to-image generation is a growing area, and the benchmark's clear public dataset and tools should attract attention from T2I and multilingual AI researchers."},"evaluationMode":"public_reusable","publishers":[{"name":"RISys Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/RISys-Lab/LingT2I","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_0e44f267d3c86725","familyId":"catalog_family_0e44f267d3c86725","name":"Liquid Extract F1","oneLine":"A display-only Liquid AI extraction metric measuring field-name agreement between requested schema fields and extracted JSON fields.","description":"A display-only Liquid AI extraction metric measuring field-name agreement between requested schema fields and extracted JSON fields.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0e44f267d3c86725"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/liquidextractschemaf1"}],"catalogSources":[{"catalog":"benchlm","sourceId":"liquidExtractSchemaF1","url":"https://benchlm.ai/benchmarks/liquidextractschemaf1","paperUrl":"https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract","year":"2026","fullName":"Liquid image-to-JSON extraction schema consistency F1","format":"Schema field F1","tasks":"Image-to-JSON extraction","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_7470719af1806d98","familyId":"catalog_family_7470719af1806d98","name":"Liquid Extract JSON Validity","oneLine":"A display-only Liquid AI extraction metric measuring the share of image-to-JSON outputs that parse as strict JSON.","description":"A display-only Liquid AI extraction metric measuring the share of image-to-JSON outputs that parse as strict JSON.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7470719af1806d98"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/liquidextractjsonvalidity"}],"catalogSources":[{"catalog":"benchlm","sourceId":"liquidExtractJsonValidity","url":"https://benchlm.ai/benchmarks/liquidextractjsonvalidity","paperUrl":"https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract","year":"2026","fullName":"Liquid image-to-JSON extraction JSON validity","format":"Strict JSON parseability rate","tasks":"Image-to-JSON extraction","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_050e8ebba31c6ed8","familyId":"catalog_family_050e8ebba31c6ed8","name":"Liquid Extract VLM Judge","oneLine":"A display-only Liquid AI extraction metric measuring judged agreement between extracted values and the source image.","description":"A display-only Liquid AI extraction metric measuring judged agreement between extracted values and the source image.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_050e8ebba31c6ed8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/liquidextractvlmjudge"}],"catalogSources":[{"catalog":"benchlm","sourceId":"liquidExtractVlmJudge","url":"https://benchlm.ai/benchmarks/liquidextractvlmjudge","paperUrl":"https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract","year":"2026","fullName":"Liquid image-to-JSON extraction VLM judge score","format":"VLM-judged extraction accuracy","tasks":"Image-to-JSON extraction","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_8bf0602e53541de6","familyId":"catalog_family_8bf0602e53541de6","name":"LisanBench","oneLine":"A word-chain reasoning benchmark that tests planning, recall, constraint following, and vocabulary depth by asking models to extend non-repeating edit-distance-1 chains.","description":"A word-chain reasoning benchmark that tests planning, recall, constraint following, and vocabulary depth by asking models to extend non-repeating edit-distance-1 chains.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://lisanbench.com/?tab=about","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8bf0602e53541de6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/lisanbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"lisanBench","url":"https://benchlm.ai/benchmarks/lisanbench","paperUrl":"https://lisanbench.com/?tab=about","year":"2026","fullName":"LisanBench","format":"Difficulty-weighted word-chain reasoning","tasks":"50 starting words × 3 trials","successorKey":null}],"catalogCategories":["reasoning"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_what-proves-you-wrong-benchmarking-languag_17c27305","familyId":"bmf_1fd2a8dfdbdd","name":"Lit2Test","oneLine":"A benchmark for falsifiable research ideation, using a six-field contract centered on a falsifying outcome, with pairwise comparisons of proposals from four models.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":0.45,"links":{"report":"http://arxiv.org/abs/2608.22948v1","pdf":"https://arxiv.org/pdf/2608.22948v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce a benchmark that carries a proposal from Literature to Test: the Lit2Test benchmark centers on a six-field contract organized around a falsifying outcome, so that every proposal precommits the observation that would prove it wrong, making its quality decidable in the first place rather than merely arguable.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22948"},"ranking":{"today":{"score":52,"rank":4,"coverage":0.4,"confidence":"Low"},"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark for falsifiable research ideation, using a six-field contract centered on a falsifying outcome, with pairwise comparisons of proposals from four models.","whyItMatters":"Introduces a decidable evaluation contract for research proposals, moving beyond free-form judging and enabling reliable comparison of model-generated research ideas.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"75d0b247e1ccacafb32f04445a1aef8bd0e7cc92519ae8949023be33dc433863"},"motivation":"Large language models are increasingly used to propose research ideas, yet the prevailing ways of judging such ideas supply no shared decision rule: free-form judging sways with style and position, and scoring against a later paper rewards recovery of one realized trajectory.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with a clear scoring protocol, public release statement, and prospective comparison of frontier models.","canonicalNameSource":"abstract","canonicalNameEvidence":"the Lit2Test benchmark centers on a six-field contract organized around a falsifying outcome"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22948v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":55,"confidence":"Low","horizon":"7d","reason":"Broad relevance to AI research evaluation and LLM use; anticipated interest but unverified artifact access."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_littraceqa_955ebe65","familyId":"bmf_c0a9eaf07d4f","name":"LitTraceQA","oneLine":"LitTraceQA evaluates literature-grounded question answering over scientific papers, requiring systems to return paper IDs, evidence locations, and answers in multiple formats. The public split includes 55 examples with gold annotations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07370","pdf":"https://arxiv.org/pdf/2608.07370","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07370"},"evidence":{"snippet":"We present LitTraceQA, a benchmark for literature-grounded question answering over scientific papers.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07370"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LitTraceQA evaluates literature-grounded question answering over scientific papers, requiring systems to return paper IDs, evidence locations, and answers in multiple formats. The public split includes 55 examples with gold annotations.","whyItMatters":"The benchmark fills a gap in evaluating verifiable scientific QA, separating retrieval, grounding, and answer accuracy, providing a reusable testbed for systems that produce evidence-backed responses.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d95068159e4bc1b52bc5a43f15d065ca9cd34b00b931a053d1205a82e885ffd5"},"motivation":"Scientific literature is increasingly used as a knowledge source for language models, retrieval-augmented generation systems, and research assistants, but answering research questions from papers requires more than fluent generation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07370","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_858362768f842e9d","familyId":"catalog_family_858362768f842e9d","name":"LiveBench","oneLine":"LiveBench is a challenging, contamination-limited LLM benchmark that addresses test set contamination by releasing new questions monthly based on recently-released datasets, arXiv papers, news articles, and IMDb movie synopses. It comprises tasks across math, coding, reasoning, language, instruction following, and data analysis with verifiable, objective ground-truth answers.","description":"LiveBench is a challenging, contamination-limited LLM benchmark that addresses test set contamination by releasing new questions monthly based on recently-released datasets, arXiv papers, news articles, and IMDb movie synopses. It comprises tasks across math, coding, reasoning, language, instruction following, and data analysis with verifiable, objective ground-truth answers.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External","Math","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://livebench.ai/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_858362768f842e9d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/livebench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/livebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"liveBench","url":"https://benchlm.ai/benchmarks/livebench","paperUrl":"https://livebench.ai/","year":"2024","fullName":"LiveBench","format":"Mean of category averages","tasks":"23 objective tasks across 7 categories","successorKey":null},{"catalog":"llm-stats","sourceId":"livebench","url":"https://llm-stats.com/benchmarks/livebench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["external","math","reasoning","general"],"catalogModelCount":38,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_7869f24ecad81765","familyId":"catalog_family_7869f24ecad81765","name":"LiveBench 20241125","oneLine":"LiveBench is a challenging, contamination-limited LLM benchmark that addresses test set contamination by releasing new questions monthly based on recently-released datasets, arXiv papers, news articles, and IMDb movie synopses. It comprises tasks across math, coding, reasoning, language, instruction following, and data analysis with verifiable, objective ground-truth answers.","description":"LiveBench is a challenging, contamination-limited LLM benchmark that addresses test set contamination by releasing new questions monthly based on recently-released datasets, arXiv papers, news articles, and IMDb movie synopses. It comprises tasks across math, coding, reasoning, language, instruction following, and data analysis with verifiable, objective ground-truth answers.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/livebench-20241125","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7869f24ecad81765"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/livebench-20241125"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"livebench-20241125","url":"https://llm-stats.com/benchmarks/livebench-20241125","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning","general"],"catalogModelCount":14,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_livecodebench_fa145f1d","familyId":"bmf_b33aa0adf0e0","name":"LiveCodeBench","oneLine":"CurveShift analyzes progress on LiveCodeBench, releasing a difficulty panel of 66 models and 1,055 problems.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00355","pdf":"https://arxiv.org/pdf/2608.00355","project":null,"code":"https://github.com/harvenstar/CurveShift","data":null,"hfPaper":"https://huggingface.co/papers/2608.00355"},"evidence":{"snippet":"We release the LiveCodeBench Difficulty Panel (66 dated models x 1,055 problems) and our analysis code.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00355"},"ranking":{"90d":{"score":28,"rank":269,"coverage":0.55,"confidence":"Low"}},"description":"CurveShift analyzes progress on LiveCodeBench, releasing a difficulty panel of 66 models and 1,055 problems.","whyItMatters":"Provides insights into whether progress is scalar or shaped by task difficulty, but does not define a new evaluation benchmark.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ae80d13489ec097ddf97a43759b4f16ca26250966a72f9a4e69162f9ad4704d9"},"motivation":"Progress in large language models is often summarized using a single scalar measure, such as a time horizon, a latent ability estimate, or an aggregate benchmark score.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00355","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"liveCodeBench","url":"https://benchlm.ai/benchmarks/livecodebench","paperUrl":"https://arxiv.org/abs/2403.07974","year":"2024","fullName":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","format":"Competitive-programming evaluation","tasks":"Continuously updated contest problems","successorKey":null},{"catalog":"llm-stats","sourceId":"livecodebench","url":"https://llm-stats.com/benchmarks/livecodebench","datasetSlug":"livecodebench","versionCount":4,"subsetCount":1,"rowCount":1055,"updatedAt":"2026-06-11T15:38:43.238317+00:00","community":true}],"catalogCategories":["coding","reasoning","general","code"],"catalogModelCount":75,"catalogStarCount":4},{"id":"lib_livecodebench","familyId":"family_livecodebench","name":"LiveCodeBench","oneLine":"Established benchmark family · Code & Software.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code & Software"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2403.07974","pdf":null,"project":"https://livecodebench.github.io/","code":"https://github.com/LiveCodeBench/LiveCodeBench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_livecodebench"},"ranking":{},"recordType":"family","aliases":["LCB"],"sourceAttribution":[{"role":"official-project","url":"https://livecodebench.github.io/"}],"adoptionRefs":["openai-gpt5","google-gemini25","deepseek-v3"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"},{"sourceId":"deepseek-v3","url":"https://github.com/deepseek-ai/DeepSeek-V3","provider":"DeepSeek","model":"DeepSeek-V3"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_fef7ee9f6b47c7fd","familyId":"catalog_family_fef7ee9f6b47c7fd","name":"LiveCodeBench Pass@1-COT","oneLine":"This lane contains DeepSeek's LiveCodeBench Pass@1-COT results. The explicit metric and prompting label keeps them separate from generic and version-specific LiveCodeBench rows.","description":"This lane contains DeepSeek's LiveCodeBench Pass@1-COT results. The explicit metric and prompting label keeps them separate from generic and version-specific LiveCodeBench rows.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/resolve/main/DeepSeek_V4.pdf?download=true","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fef7ee9f6b47c7fd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/livecodebenchpass1cot"}],"catalogSources":[{"catalog":"benchlm","sourceId":"liveCodeBenchPass1Cot","url":"https://benchlm.ai/benchmarks/livecodebenchpass1cot","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/resolve/main/DeepSeek_V4.pdf?download=true","year":"2026","fullName":"LiveCodeBench Pass@1 with Chain-of-Thought","format":"Pass@1-COT competitive programming results","tasks":"DeepSeek-V4 report evaluation window","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_87b50a562a887e4a","familyId":"catalog_family_87b50a562a887e4a","name":"LiveCodeBench Pro","oneLine":"LiveCodeBench Pro is an advanced evaluation benchmark for large language models for code that uses Elo ratings to rank models based on their performance on coding tasks. It evaluates models on real-world coding problems from programming contests (LeetCode, AtCoder, CodeForces) and provides a relative ranking system where higher Elo scores indicate superior performance.","description":"LiveCodeBench Pro is an advanced evaluation benchmark for large language models for code that uses Elo ratings to rank models based on their performance on coding tasks. It evaluates models on real-world coding problems from programming contests (LeetCode, AtCoder, CodeForces) and provides a relative ranking system where higher Elo scores indicate superior performance.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Reasoning","General","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2506.11928","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_87b50a562a887e4a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/livecodebench-pro"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/livecodebench-pro"}],"catalogSources":[{"catalog":"benchlm","sourceId":"liveCodeBenchPro","url":"https://benchlm.ai/benchmarks/livecodebench-pro","paperUrl":"https://arxiv.org/abs/2506.11928","year":"2025","fullName":"LiveCodeBench Pro","format":"Competitive programming","tasks":"Quarter-specific contest programming sets","successorKey":null},{"catalog":"llm-stats","sourceId":"livecodebench-pro","url":"https://llm-stats.com/benchmarks/livecodebench-pro","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","reasoning","general","code"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_ad37284f83ab5ddb","familyId":"catalog_family_ad37284f83ab5ddb","name":"LiveCodeBench v5","oneLine":"LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.","description":"LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://github.com/LiveCodeBench/LiveCodeBench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ad37284f83ab5ddb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/livecodebench-v5"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/livecodebench-v5"}],"catalogSources":[{"catalog":"benchlm","sourceId":"liveCodeBenchV5","url":"https://benchlm.ai/benchmarks/livecodebench-v5","paperUrl":"https://github.com/LiveCodeBench/LiveCodeBench","year":"2025","fullName":"LiveCodeBench v5","format":"Provider-published v5 competitive programming result","tasks":"July 2024 to May 2025 release window","successorKey":null},{"catalog":"llm-stats","sourceId":"livecodebench-v5","url":"https://llm-stats.com/benchmarks/livecodebench-v5","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","reasoning","general"],"catalogModelCount":9,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_5140b427ae8edecc","familyId":"catalog_family_5140b427ae8edecc","name":"LiveCodeBench v5 24.12-25.2","oneLine":"LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.","description":"LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/livecodebench-v5-24.12-25.2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5140b427ae8edecc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/livecodebench-v5-24.12-25.2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"livecodebench-v5-24.12-25.2","url":"https://llm-stats.com/benchmarks/livecodebench-v5-24.12-25.2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_92b6f9854a975521","familyId":"catalog_family_92b6f9854a975521","name":"LiveCodeBench v6","oneLine":"LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.","description":"LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://github.com/LiveCodeBench/LiveCodeBench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_92b6f9854a975521"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/livecodebench-v6"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/livecodebench-v6"}],"catalogSources":[{"catalog":"benchlm","sourceId":"liveCodeBenchV6","url":"https://benchlm.ai/benchmarks/livecodebench-v6","paperUrl":"https://github.com/LiveCodeBench/LiveCodeBench","year":"2026","fullName":"LiveCodeBench v6","format":"Provider-published v6 competitive programming results","tasks":"Fresh programming problems","successorKey":null},{"catalog":"llm-stats","sourceId":"livecodebench-v6","url":"https://llm-stats.com/benchmarks/livecodebench-v6","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","reasoning","general"],"catalogModelCount":58,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_d0a8b6d9c573e736","familyId":"catalog_family_d0a8b6d9c573e736","name":"LiveCodeBench(01-09)","oneLine":"LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.","description":"LiveCodeBench is a holistic and contamination-free evaluation benchmark for large language models for code. It continuously collects new problems from programming contests (LeetCode, AtCoder, CodeForces) and evaluates four different scenarios: code generation, self-repair, code execution, and test output prediction. Problems are annotated with release dates to enable evaluation on unseen problems released after a model's training cutoff.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/livecodebench(01-09)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d0a8b6d9c573e736"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/livecodebench(01-09)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"livecodebench(01-09)","url":"https://llm-stats.com/benchmarks/livecodebench(01-09)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_livehouse-ts_9c17ef53","familyId":"bmf_4ac789a70586","name":"LiveHouse-TS","oneLine":"Continuously evaluates time-series foundation models on future observations from live cross-domain data streams.","area":"Language & Knowledge","applicationDomains":["Finance & Economics","Science & Research","Transport & Logistics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":["Time-series forecasting","Temporal robustness","Distribution shift"],"topics":["cs.AI"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.17299","pdf":"https://arxiv.org/pdf/2608.17299","project":"https://huggingface.co/spaces/CityMindDev/LiveHouse-TS","code":"https://github.com/zhouziyu02/TS-Live","data":"https://huggingface.co/spaces/CityMindDev/LiveHouse-TS","hfPaper":"https://huggingface.co/papers/2608.17299"},"evidence":{"snippet":"To bridge this gap, we introduce LiveHouse-TS, the first open-world living benchmark infrastructure for TSFMs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17299"},"ranking":{"30d":{"score":23,"rank":148,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":352,"coverage":0.55,"confidence":"Low"}},"description":"LiveHouse-TS is a living benchmark for time series foundation models, evaluating models prequentially on real future data across 11 domains and 17 datasets. It shifts benchmarking from snapshot accuracy to continuous temporal validity.","whyItMatters":"Addresses the limitation of static benchmarks for time series models, providing a framework to assess model robustness under distribution shifts and long-term ranking stability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"24f5f3a0203624d8edded8c57b3e2cdf654c09f127ae57d13c047ce520fbc867"},"motivation":"Time Series Foundation Models (TSFMs) have recently emerged as a highly promising paradigm for cross-domain zero-shot forecasting.","constructionDetail":"LiveHouse-TS uses newly arriving observations from 17 datasets so forecasting models are evaluated under real distribution shift rather than a frozen test set.","detail":{"taskBreakdown":["Weather","Air quality","Energy","Hydrology","Ocean","Traffic","Finance","Web attention","Macroeconomics","News events","Disaster events"],"protocol":{"tasks":"Continuous forecasting over 17 datasets from 15 sources and 11 domains","primaryMetric":"RMSE and CRPS, summarized by Average Rank, Win Rate and Elo","version":"Living benchmark"},"leaderboardUrl":"https://huggingface.co/spaces/CityMindDev/LiveHouse-TS","submissionUrl":"https://github.com/zhouziyu02/TS-Live/issues/new?template=community-model.yml"},"curation":{"state":"source-reviewed","reviewedAt":"2026-08-20","sources":["https://arxiv.org/abs/2608.17299","https://arxiv.org/html/2608.17299","https://github.com/zhouziyu02/TS-Live","https://huggingface.co/spaces/CityMindDev/LiveHouse-TS"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17299","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"score_submission","availability":{"paperStatus":"available","githubStatus":"available","hfDatasetStatus":"available","evaluatorStatus":"available","submissionStatus":"available"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"bm_livek12bench_56b774f5","familyId":"bmf_80ca011e62fa","name":"LiveK12Bench","oneLine":"Dynamic benchmark for multimodal reasoning on high school exam questions in math, physics, chemistry, and biology, with a mock exam evaluation scheme.","area":"Multimodal","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals"],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26781","pdf":"https://arxiv.org/pdf/2605.26781","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.26781"},"evidence":{"snippet":"To address these issues, we introduce LiveK12Bench, a dynamic, holistic, multi-disciplinary benchmark designed to evaluate the reasoning abilities of LMMs in realistic examination scenarios.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26781"},"ranking":{},"description":"Dynamic benchmark for multimodal reasoning on high school exam questions in math, physics, chemistry, and biology, with a mock exam evaluation scheme.","whyItMatters":"Static benchmarks are prone to contamination and fail to reflect real exam constraints. LiveK12Bench provides a growing, realistic evaluation for educational readiness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a89c4dd2104625b9a43335eb8c12537263c7c9dc148e5fff33ed13594da1392a"},"motivation":"Advanced Large Multimodal Models (LMMs) have demonstrated impressive performance in K-12 reasoning tasks, exhibiting great promise as intelligent tutors.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26781","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LiveK12Bench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2605.26781","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_03f16cb9366e9fe2","familyId":"catalog_family_03f16cb9366e9fe2","name":"LiveMathematicianBench","oneLine":"LiveMathematicianBench evaluates research-level mathematical reasoning on continuously refreshed, contamination-resistant problems.","description":"LiveMathematicianBench evaluates research-level mathematical reasoning on continuously refreshed, contamination-resistant problems.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/livemathematicianbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_03f16cb9366e9fe2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/livemathematicianbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"livemathematicianbench","url":"https://llm-stats.com/benchmarks/livemathematicianbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_5bdd7249a566942c","familyId":"catalog_family_5bdd7249a566942c","name":"LiveSports-3K","oneLine":"LiveSports-3K evaluates fine-grained understanding and commentary of live sports video.","description":"LiveSports-3K evaluates fine-grained understanding and commentary of live sports video.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/livesports-3k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5bdd7249a566942c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/livesports-3k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"livesports-3k","url":"https://llm-stats.com/benchmarks/livesports-3k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","video","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_8d71d0c357112c5d","familyId":"catalog_family_8d71d0c357112c5d","name":"LiveSQLBench","oneLine":"LiveSQLBench evaluates models on generating correct SQL queries against live PostgreSQL databases. The LiveSQLBench-Base-Full v1 dataset contains 600 questions across 22 PostgreSQL databases, testing real-world database reasoning and query generation.","description":"LiveSQLBench evaluates models on generating correct SQL queries against live PostgreSQL databases. The LiveSQLBench-Base-Full v1 dataset contains 600 questions across 22 PostgreSQL databases, testing real-world database reasoning and query generation.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/livesqlbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8d71d0c357112c5d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/livesqlbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"livesqlbench","url":"https://llm-stats.com/benchmarks/livesqlbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_ll-bench_df7573ac","familyId":"bmf_e383e91589a5","name":"LL-Bench","oneLine":"LL-Bench evaluates large-scale generative models on 16 low-level vision tasks using 2,469 real-world degraded images, with human preference and quality score annotations for model outputs.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02535","pdf":"https://arxiv.org/pdf/2606.02535","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02535"},"evidence":{"snippet":"To address this gap, we introduce \\textbf{LL-Bench}, a comprehensive \\textbf{Benchmark} for evaluating the capabilities of large-scale generative models on \\textbf{L}ow-\\textbf{L}evel vision tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02535"},"ranking":{},"description":"LL-Bench evaluates large-scale generative models on 16 low-level vision tasks using 2,469 real-world degraded images, with human preference and quality score annotations for model outputs.","whyItMatters":"Existing low-level vision benchmarks often focus on conventional models and lack alignment with human perception. LL-Bench provides a standardized evaluation suite to compare generative and restoration models on pixel-level tasks, supporting quality assessment and model selection.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7ba0d380642efecde1cd833658bd564758243f39f900f18ad4067939ea0eda2d"},"motivation":"Large-scale generative models have demonstrated remarkable capabilities across image generation and editing tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02535","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_llm-summarization-benchmark-ptbr_c5bebc73","familyId":"bmf_d1dd765edb10","name":"LLM Summarization Benchmark — Brazilian Portuguese","oneLine":"Evaluates reference-faithful summarization of a technical report in Brazilian Portuguese across 9 local LLMs, using a pre-registered rubric with six runs per model and reported scores, timing, and speed.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-21","firstSeenAt":"2026-08-22","recognitionConfidence":0.4,"links":{"report":"https://github.com/gbbarra/llm-summarization-benchmark-ptbr","pdf":null,"project":"https://doi.org/10.5281/zenodo.22042502","code":"https://github.com/gbbarra/llm-summarization-benchmark-ptbr","data":"https://zenodo.org/badge/DOI/10.5281/zenodo.22042502.svg","hfPaper":null},"evidence":{"snippet":"llm-summarization-benchmark-ptbr # LLM Summarization Benchmark — Brazilian Portuguese [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.22042502.svg)](https://doi.org/10.5281/zenodo.22042502) *Six runs per model, a pre-registered rubric, and a trap-laden source text: which local LLMs summarize a technical report with strict faithfulness to the reference — and how fast — on a mini PC with no discrete GPU?* A fully reproducible benchmark of **reference-faithful summarization** across 9 local LLM","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:gbbarra/llm-summarization-benchmark-ptbr"},"ranking":{"30d":{"score":23,"rank":132,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":336,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates reference-faithful summarization of a technical report in Brazilian Portuguese across 9 local LLMs, using a pre-registered rubric with six runs per model and reported scores, timing, and speed.","whyItMatters":"Provides a scarce multilingual evaluation resource for Portuguese summarization with detailed protocol and raw outputs, enabling reproducibility and comparison of local model faithfulness and efficiency.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"af5dc59bf039ba03abedd2090646bbd9f3d58d84fe011d581b56368e2464457b"},"motivation":"llm-summarization-benchmark-ptbr # LLM Summarization Benchmark — Brazilian Portuguese [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.22042502.svg)](https://doi.org/10.5281/zenodo.22042502) *Six runs per model, a pre-registered rubric, and a trap-laden source text: which local LLMs summarize a technical report with strict faithfulness to the reference — and how fast — on a mini PC with no discrete GPU?* A fully…","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/gbbarra/llm-summarization-benchmark-ptbr","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-22T19:19:35.289219Z"},"attentionForecast":{"score":52,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses a niche but growing area of multilingual LLM evaluation with detailed artifacts, likely attracting moderate interest from researchers focused on non-English benchmarks."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_llm-finetuning-benchmark_efda9d1c","familyId":"bmf_663a79d7b514","name":"llm-finetuning-benchmark","oneLine":"Benchmarks fine-tuning strategies for LLMs using Financial PhraseBank and Qwen 2.5-0.5B.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-24","recognitionConfidence":0.4,"links":{"report":"https://github.com/Ishant-005/llm-finetuning-benchmark","pdf":null,"project":null,"code":"https://github.com/Ishant-005/llm-finetuning-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"llm-finetuning-benchmark Benchmarked different fine tuning strategies like full ft, LoRA, QLoRA.","reasonCodes":["discovered via github","benchmark term in abstract","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:ishant-005/llm-finetuning-benchmark"},"ranking":{"30d":{"score":23,"rank":120,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":324,"coverage":0.55,"confidence":"Low"}},"description":"Benchmarks fine-tuning strategies for LLMs using Financial PhraseBank and Qwen 2.5-0.5B.","whyItMatters":"The project does not provide a reusable benchmark protocol or public comparison path.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"27f229f3ead4feaad9b48d42829d73e13ba8468edfa967cbed9a31bc7d6cd476"},"motivation":"llm-finetuning-benchmark Benchmarked different fine tuning strategies like full ft, LoRA, QLoRA.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The repository is a one-off experiment rather than a formal benchmark, lacking a stable scoring contract or reusable protocol."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/Ishant-005/llm-finetuning-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-24T07:40:52.396154Z"},"attentionForecast":{"score":0,"confidence":"Low","horizon":"7d","reason":"Excluded because it does not meet benchmark criteria."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_llm-free-benchmark_d7c7d60c","familyId":"bmf_aa59460ae398","name":"llm-free-benchmark","oneLine":"Runs a fixed battery of seven typed tasks against free-tier LLMs, grades each answer programmatically on a 0-5 scale, and reports rankings with per-task scores and latency.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-21","firstSeenAt":"2026-08-22","recognitionConfidence":0.4,"links":{"report":"https://github.com/Miolonixc/llm-free-benchmark","pdf":null,"project":"https://openrouter.ai/keys","code":"https://github.com/Miolonixc/llm-free-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"llm-free-benchmark Dependency-free benchmark for free-tier LLMs on OpenRouter.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:miolonixc/llm-free-benchmark"},"ranking":{"30d":{"score":23,"rank":133,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":337,"coverage":0.55,"confidence":"Low"}},"description":"Runs a fixed battery of seven typed tasks against free-tier LLMs, grades each answer programmatically on a 0-5 scale, and reports rankings with per-task scores and latency.","whyItMatters":"Provides a simple, dependency-free way to compare free LLM offerings, addressing rate-limit artifacts and enabling ongoing monitoring of model quality.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"26a2444e8c6aa5628c90b2723571f1427966207210ba5d960a3f4c16e02d0cf5"},"motivation":"llm-free-benchmark Dependency-free benchmark for free-tier LLMs on OpenRouter.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/Miolonixc/llm-free-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-22T19:19:35.289219Z"},"attentionForecast":{"score":38,"confidence":"Low","horizon":"7d","reason":"The niche focus on free-tier models and minimal task battery may limit immediate attention, but its accessibility could attract some users."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_llmtabbench_435990cf","familyId":"bmf_1ca641025f84","name":"LLMTabBench","oneLine":"LLMTabBench evaluates LLMs on binary tabular classification in zero- and few-shot settings, using real-world and controlled synthetic datasets to study the effect of task descriptions and examples on performance.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24417","pdf":"https://arxiv.org/pdf/2605.24417","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24417"},"evidence":{"snippet":"We introduce LLMTabBench, a benchmark for evaluating LLMs on tabular classification under low-data conditions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24417"},"ranking":{},"description":"LLMTabBench evaluates LLMs on binary tabular classification in zero- and few-shot settings, using real-world and controlled synthetic datasets to study the effect of task descriptions and examples on performance.","whyItMatters":"The benchmark addresses the gap in understanding how LLMs perform on tabular data under low-data regimes, which can inform their deployment in data-scarce applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"aaa39281bc980dd951a3b446a8ee3ba90e342a47e43bbe9a092d65609af20f76"},"motivation":"Supervised classification on tabular data remains a central machine learning task, but its dependence on large labeled datasets limits its applicability in data-scarce settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24417","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_llvm-bench_e576b85e","familyId":"bmf_efd956bb75ce","name":"LLVM-Bench","oneLine":"LLVM-Bench evaluates LLMs on resolving LLVM compiler issues with 423 real-world tasks, using an automated evaluation platform.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.00700","pdf":"https://arxiv.org/pdf/2607.00700","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.00700"},"evidence":{"snippet":"To address this gap, we introduce LLVM-Bench, the first large-scale benchmark for LLVM issue resolution, containing 423 real-world, validated tasks collected from the LLVM project.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.00700"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LLVM-Bench evaluates LLMs on resolving LLVM compiler issues with 423 real-world tasks, using an automated evaluation platform.","whyItMatters":"Provides a large-scale benchmark for system-level compiler issue resolution, addressing a gap in LLM evaluation for complex software engineering tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b1f821a5a5dbd4c5977eb5fb437718f38e0fe2fe36d66dff4608d75c857b5ebb"},"motivation":"LLVM is a widely used compiler infrastructure whose scale and complexity make issue resolution labor-intensive and challenging.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.00700","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_a43419ec57910ff1","familyId":"catalog_family_a43419ec57910ff1","name":"LMArena Text Leaderboard","oneLine":"LMArena Text Leaderboard is a blind human preference evaluation benchmark that ranks models based on pairwise comparisons in real-world conversations. The leaderboard uses Elo ratings computed from user preferences in head-to-head model battles, providing a comprehensive measure of overall model capability and style.","description":"LMArena Text Leaderboard is a blind human preference evaluation benchmark that ranks models based on pairwise comparisons in real-world conversations. The leaderboard uses Elo ratings computed from user preferences in head-to-head model battles, providing a comprehensive measure of overall model capability and style.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/lmarena-text","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a43419ec57910ff1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/lmarena-text"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"lmarena-text","url":"https://llm-stats.com/benchmarks/lmarena-text","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_c158bb91f29e76ad","familyId":"catalog_family_c158bb91f29e76ad","name":"LOCA-Bench (256k)","oneLine":"LOCA-Bench is a long-context agentic benchmark. The 256k variant evaluates agents using the official ReAct mode with an environment description length of 256k tokens, measuring how well models reason and act over very long contexts.","description":"LOCA-Bench is a long-context agentic benchmark. The 256k variant evaluates agents using the official ReAct mode with an environment description length of 256k tokens, measuring how well models reason and act over very long contexts.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/loca-bench-256k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c158bb91f29e76ad"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/loca-bench-256k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"loca-bench-256k","url":"https://llm-stats.com/benchmarks/loca-bench-256k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_logdx-ci_ff295ed3","familyId":"bmf_cb2e3cced8f6","name":"LogDx-CI","oneLine":"Benchmark evaluating 11 log reduction tools on 35 GitHub Actions failure cases, scored by 3 LLM debugger families.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.28876","pdf":"https://arxiv.org/pdf/2605.28876","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28876"},"evidence":{"snippet":"We introduce LogDx-CI, a benchmark that compares 11 context-reduction tools (raw, tail, grep, three RTK modes, two real LLM map-reduce summarizers, three hybrid routers) on 35 real GitHub Actions failure cases, scored by 3 LLM debugger families (Claude Haiku 4.5, Claude Sonnet 4.6, OpenAI gpt-5-mini) plus a Sonnet 4.6 tool-using agent.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28876"},"ranking":{},"description":"Benchmark evaluating 11 log reduction tools on 35 GitHub Actions failure cases, scored by 3 LLM debugger families.","whyItMatters":"No public comparison existed for which log reductions preserve diagnostic evidence for LLM-based root-cause diagnosis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a17083f3be20a107653db3c4ee28e0dbe9d3f937bc5450b997dab3955f09ee2a"},"motivation":"CI failure logs are large (median 5k lines, max 200k in this corpus) and noisy.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28876","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LogDx-CI Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2605.28876","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_lomevqa_dd681f2a","familyId":"bmf_1b89ee4a1e86","name":"LoMeVQA","oneLine":"LoMeVQA is a benchmark for longitudinal medical visual question answering, with 206K VQA pairs across five tasks: progress classification, progress description, progress report generation, differential region grounding, and differential region description.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27806","pdf":"https://arxiv.org/pdf/2607.27806","project":null,"code":"https://github.com/pepperbubble/LoMeVQA","data":null,"hfPaper":"https://huggingface.co/papers/2607.27806"},"evidence":{"snippet":"To fill this gap, we propose LoMeVQA, a comprehensive benchmark consisting of 206K longitudinal visual question answering (VQA) pairs for temporal medical image analysis.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27806"},"ranking":{"90d":{"score":23,"rank":304,"coverage":0.7,"confidence":"Medium"}},"description":"LoMeVQA is a benchmark for longitudinal medical visual question answering, with 206K VQA pairs across five tasks: progress classification, progress description, progress report generation, differential region grounding, and differential region description.","whyItMatters":"Longitudinal medical reasoning is underexplored in MLLMs, and this benchmark provides a comprehensive testbed. It reveals limitations in temporal reasoning and supports future improvements in medical AI.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"aa1e98156bb41f79638f8dde30a86dd02a23d36af052749bfc0c7b822d960fa5"},"motivation":"In clinical practice, patients often undergo multiple imaging examinations over successive visits, yielding longitudinal data.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27806","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LoMeVQA Project Team","organizationType":"academic-lab","sourceUrl":"https://github.com/pepperbubble/LoMeVQA","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_long-horizon-terminal-bench_593d14a6","familyId":"bmf_0d2d7122f15f","name":"Long-Horizon-Terminal-Bench","oneLine":"Evaluates long-horizon terminal tasks in a containerized environment with hidden verifiers and dense reward grading across 46 tasks and nine categories.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Multimodal","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.08964","pdf":"https://arxiv.org/pdf/2607.08964","project":null,"code":"https://github.com/zli12321/LHTB","data":"https://huggingface.co/datasets/IntelligenceLab/Long-Horizon-Terminal-Bench","hfPaper":"https://huggingface.co/papers/2607.08964"},"evidence":{"snippet":"We introduce Long-Horizon-Terminal-Bench, a terminal benchmark of 46 long-horizon tasks spanning nine categories, including experiment reproduction, software engineering, multimodal analysis, interactive games, and scientific computing.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":77,"hfDailySubmittedAt":"2026-07-13T00:00:00.000Z","githubStars":700,"githubScope":"benchmark_repo","hfDatasetDownloads":8710,"hfDatasetLikes":132},"source":{"type":"arxiv","id":"2607.08964"},"ranking":{"90d":{"score":83,"rank":2,"coverage":1.0,"confidence":"High","datasetDownloadRank":4,"datasetRankPopulation":66}},"description":"Evaluates long-horizon terminal tasks in a containerized environment with hidden verifiers and dense reward grading across 46 tasks and nine categories.","whyItMatters":"Provides a more demanding evaluation for agentic long-horizon planning and partial credit, addressing gaps in existing terminal benchmarks that only measure final outcomes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"77d3230de58f2233270e143a2d5ffcbf256c0414dba28bafbacab8bdb5e5148b"},"motivation":"AI agents have become capable of autonomously completing short, well-specified tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"source-reviewed","reviewedAt":"2026-08-30","sources":["https://arxiv.org/abs/2607.08964","https://github.com/zli12321/LHTB","https://huggingface.co/datasets/IntelligenceLab/Long-Horizon-Terminal-Bench"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.08964","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_longav-compass_f7e880eb","familyId":"bmf_67cc07e08a0c","name":"LongAV-Compass","oneLine":"LongAV-Compass evaluates minute-scale audio-visual generation across text-to-audio-video, image-to-audio-video, and video-to-audio-video tasks. It includes 284 curated test cases and an evaluation framework combining multimodal metrics with MLLM-assisted assessment across 20+ dimensions covering segment quality, cross-segment consistency, narrative coherence, semantic alignment, and audiovisual synchronization.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26244","pdf":"https://arxiv.org/pdf/2605.26244","project":null,"code":"https://github.com/pkucs-Ltf/LongAV-Compass","data":null,"hfPaper":"https://huggingface.co/papers/2605.26244"},"evidence":{"snippet":"To bridge this gap, we introduce LongAV-Compass, a systematic benchmark for minute-long audio-visual generation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":38,"hfDailySubmittedAt":null,"githubStars":17,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26244"},"ranking":{},"description":"LongAV-Compass evaluates minute-scale audio-visual generation across text-to-audio-video, image-to-audio-video, and video-to-audio-video tasks. It includes 284 curated test cases and an evaluation framework combining multimodal metrics with MLLM-assisted assessment across 20+ dimensions covering segment quality, cross-segment consistency, narrative coherence, semantic alignment, and audiovisual synchronization.","whyItMatters":"Existing audio-visual evaluation is limited to short clips, leaving a gap for minute-scale generation. LongAV-Compass provides a standardized protocol for diagnosing degradation in identity consistency, narrative coherence, and alignment over long horizons, supporting comparison of long-form generation models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"903d8a598d71a3ad64fe4c5d7fe6a2514e7320ca4ad22b1e9d5c1061ad0eb1d6"},"motivation":"Audio-visual generation is rapidly advancing from short clips to minute-long content, while existing evaluation protocols remain largely confined to short-form settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26244","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"PKU CS LongAV Group","organizationType":"academic-lab","sourceUrl":"https://github.com/pkucs-Ltf/LongAV-Compass","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"lib_longbench_v2","familyId":"family_longbench","name":"LongBench v2","oneLine":"Established benchmark variant · Long Context.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2412.15204","pdf":null,"project":"https://longbench2.github.io/","code":"https://github.com/THUDM/LongBench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_longbench_v2"},"ranking":{},"recordType":"variant","aliases":["LongBench V2"],"sourceAttribution":[{"role":"official-project","url":"https://longbench2.github.io/"}],"adoptionRefs":["google-gemini25"],"modelReportReferences":[{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"variantOfExternal":"LongBench","capabilityGroups":["Long Context & Memory"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"longBenchV2","url":"https://benchlm.ai/benchmarks/longbench-v2","paperUrl":"https://arxiv.org/abs/2412.15204","year":"2025","fullName":"LongBench v2","format":"Extended-context retrieval and reasoning","tasks":"Long-context tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"longbench-v2","url":"https://llm-stats.com/benchmarks/longbench-v2","datasetSlug":"longbench-v2","versionCount":7,"subsetCount":3,"rowCount":null,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":["reasoning","long context","structured output","general"],"catalogModelCount":17,"catalogStarCount":0},{"id":"catalog_bd5708897110de7e","familyId":"catalog_family_bd5708897110de7e","name":"LongCodeBench","oneLine":"LongCodeBench evaluates the code understanding and comprehension abilities of large language models at very long context windows, scaling up to 1M tokens. It tests whether models can reason about extensive codebases provided in a single prompt by answering multiple-choice questions about the code.","description":"LongCodeBench evaluates the code understanding and comprehension abilities of large language models at very long context windows, scaling up to 1M tokens. It tests whether models can reason about extensive codebases provided in a single prompt by answering multiple-choice questions about the code.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/longcodebench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bd5708897110de7e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/longcodebench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"longcodebench","url":"https://llm-stats.com/benchmarks/longcodebench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering","Long Context & Memory"],"domainScope":"general"},{"id":"bm_longdocbench_5cb5bcf9","familyId":"bmf_25cf6240e611","name":"LongDocBench","oneLine":"LongDocBench is a benchmark for Table-of-Contents Hierarchy Recovery and Contextual Relationship Recovery in long documents. It includes 85 real-world documents (financial reports, textbooks, academic papers) spanning 2,582 pages, with human-verified annotations for 3,937 heading nodes and 3,258 contextual relationships.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15064","pdf":"https://arxiv.org/pdf/2608.15064","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.15064"},"evidence":{"snippet":"To benchmark these two tasks, we introduce \\textsc{LongDocBench}, comprising 85 real-world financial reports, textbooks, and academic papers spanning 2,582 pages, with up to 105 pages per document.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15064"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LongDocBench is a benchmark for Table-of-Contents Hierarchy Recovery and Contextual Relationship Recovery in long documents. It includes 85 real-world documents (financial reports, textbooks, academic papers) spanning 2,582 pages, with human-verified annotations for 3,937 heading nodes and 3,258 contextual relationships.","whyItMatters":"Existing document parsing benchmarks focus on page-level tasks, leaving document-level structure recovery unevaluated. LongDocBench provides a standardized evaluation for tasks that are critical for understanding long documents, enabling comparison of parsers on hierarchy and relationship recovery.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bf1f43a4a7ff195e80e0cbf6f97d300f9114d2118aee8cde95428800039282d6"},"motivation":"Parsing visual documents into machine-readable representations is fundamental to document intelligence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15064","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LongDocBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.15064","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_longds-bench_2436ff1e","familyId":"bmf_95132b949b16","name":"LongDS-Bench","oneLine":"LongDS-Bench evaluates long-horizon, multi-turn data analysis tasks where agents must maintain, update, restore, and compose evolving analytical states. It comprises 68 tasks from real-world Kaggle notebooks spanning 2,225 turns across six domains, with an average dependency span of 11.3 turns.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30434","pdf":"https://arxiv.org/pdf/2605.30434","project":null,"code":"https://github.com/zjunlp/DataMind","data":null,"hfPaper":"https://huggingface.co/papers/2605.30434"},"evidence":{"snippet":"We introduce LongDS, a benchmark for long-horizon, multi-turn data analysis where agents must maintain, update, restore, and compose evolving analytical states.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":23,"hfDailySubmittedAt":null,"githubStars":135,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30434"},"ranking":{},"description":"LongDS-Bench evaluates long-horizon, multi-turn data analysis tasks where agents must maintain, update, restore, and compose evolving analytical states. It comprises 68 tasks from real-world Kaggle notebooks spanning 2,225 turns across six domains, with an average dependency span of 11.3 turns.","whyItMatters":"Existing benchmarks focus on isolated or short interactive tasks, leaving long-horizon analytical state management untested. This benchmark reveals a critical bottleneck in agent performance, where errors concentrate in later turns and additional interaction steps do not reliably improve accuracy, aiding development of more reliable agentic systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"aa20c40037c63cfe5d24156a4594e58fc804546ab10e82a764151d14dcbfc87c"},"motivation":"Real-world data analysis is inherently iterative, yet existing benchmarks mostly evaluate isolated or short interactive tasks, leaving agents' ability to track evolving analytical context over long horizons untested.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30434","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Zhejiang University NLP Lab (ZJUNLP)","organizationType":"academic-lab","sourceUrl":"https://github.com/zjunlp/DataMind","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_longearth-bench_e5b3d65a","familyId":"bmf_3dd2e8ddc733","name":"LongEarth-Bench","oneLine":"LongEarth-Bench evaluates vision-language models on long-horizon Earth observation reasoning, with ~120k QA samples from 117k images, sequences avg 15.14 frames (up to 30), covering 12 tasks in evolution summarization, spatial reasoning, anomaly identification, and logical prediction.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.13344","pdf":"https://arxiv.org/pdf/2608.13344","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.13344"},"evidence":{"snippet":"We introduce LongEarth-Bench, a benchmark containing approximately 120k question-answering samples derived from 117k unique images.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13344"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LongEarth-Bench evaluates vision-language models on long-horizon Earth observation reasoning, with ~120k QA samples from 117k images, sequences avg 15.14 frames (up to 30), covering 12 tasks in evolution summarization, spatial reasoning, anomaly identification, and logical prediction.","whyItMatters":"LongEarth-Bench addresses the lack of benchmarks for long-sequence Earth observation reasoning, providing a more realistic and challenging evaluation for models designed to analyze multi-stage geographic changes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a093c857c6fcc345c6995a9d82ac45ad0f3f780c327347bdce530f8ea5b81003"},"motivation":"Long-horizon Earth observation reasoning requires models to organize multi-stage geographic evolution, localize spatial changes, detect temporal anomalies, and infer future from extended image sequences.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13344","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_longegorefer_d5a8d99b","familyId":"bmf_b9bf12a8afb9","name":"LongEgoRefer","oneLine":"LongEgoRefer is a benchmark for Video Referring Expression Comprehension in long-form egocentric videos, built from Ego4D. It contains 1,498 referring expressions over videos averaging 45 minutes, requiring temporal grounding of when an occurrence happens and spatial grounding of where the object appears. The benchmark uses evaluation metrics for temporal and spatial grounding.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.02096","pdf":"https://arxiv.org/pdf/2607.02096","project":null,"code":"https://github.com/shunya-kato/LongEgoRefer","data":null,"hfPaper":"https://huggingface.co/papers/2607.02096"},"evidence":{"snippet":"To address this limitation, we introduce LongEgoRefer, a novel and challenging benchmark constructed from long-form videos in the Ego4D dataset.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02096"},"ranking":{"90d":{"score":23,"rank":375,"coverage":0.55,"confidence":"Low"}},"description":"LongEgoRefer is a benchmark for Video Referring Expression Comprehension in long-form egocentric videos, built from Ego4D. It contains 1,498 referring expressions over videos averaging 45 minutes, requiring temporal grounding of when an occurrence happens and spatial grounding of where the object appears. The benchmark uses evaluation metrics for temporal and spatial grounding.","whyItMatters":"Existing egocentric Video REC benchmarks focus on short clips, not reflecting real-world long-form recordings. This benchmark defines a demanding spatio-temporal grounding problem that tests models on sparse object occurrences and complex interactions over extended sequences.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"32c32a2217c1b77e91a22a4a2fe81eee443752aa63cc9fea7294a3cef7677b50"},"motivation":"Egocentric videos capture rich and diverse human-object interactions and have emerged as a fundamental resource for understanding human activities related to objects.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02096","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_711338389ffa8b3c","familyId":"catalog_family_711338389ffa8b3c","name":"LongFact","oneLine":"LongFact evaluates factual precision over long-form generations containing many individual claims. Each claim is extracted and verified, and the model is scored on claim-level precision, measuring whether extended responses introduce unsupported or false statements.","description":"LongFact evaluates factual precision over long-form generations containing many individual claims. Each claim is extracted and verified, and the model is scored on claim-level precision, measuring whether extended responses introduce unsupported or false statements.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Factuality","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/longfact","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_711338389ffa8b3c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/longfact"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"longfact","url":"https://llm-stats.com/benchmarks/longfact","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["factuality","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e780d7ef36edc690","familyId":"catalog_family_e780d7ef36edc690","name":"LongFact Concepts","oneLine":"LongFact is a benchmark for evaluating long-form factuality in large language models. It comprises 2,280 fact-seeking prompts spanning 38 topics, designed to test a model's ability to generate accurate, long-form responses. The benchmark uses SAFE (Search-Augmented Factuality Evaluator) to evaluate factual accuracy.","description":"LongFact is a benchmark for evaluating long-form factuality in large language models. It comprises 2,280 fact-seeking prompts spanning 38 topics, designed to test a model's ability to generate accurate, long-form responses. The benchmark uses SAFE (Search-Augmented Factuality Evaluator) to evaluate factual accuracy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/longfact-concepts","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e780d7ef36edc690"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/longfact-concepts"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"longfact-concepts","url":"https://llm-stats.com/benchmarks/longfact-concepts","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_9feb50aa67d76cd7","familyId":"catalog_family_9feb50aa67d76cd7","name":"LongFact Objects","oneLine":"LongFact is a benchmark for evaluating long-form factuality in large language models. It comprises 2,280 fact-seeking prompts spanning 38 topics, designed to test a model's ability to generate accurate, long-form responses. The benchmark uses SAFE (Search-Augmented Factuality Evaluator) to evaluate factual accuracy.","description":"LongFact is a benchmark for evaluating long-form factuality in large language models. It comprises 2,280 fact-seeking prompts spanning 38 topics, designed to test a model's ability to generate accurate, long-form responses. The benchmark uses SAFE (Search-Augmented Factuality Evaluator) to evaluate factual accuracy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/longfact-objects","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9feb50aa67d76cd7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/longfact-objects"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"longfact-objects","url":"https://llm-stats.com/benchmarks/longfact-objects","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_longjudgebench_1675bac5","familyId":"bmf_36c4ccde3d0e","name":"LongJudgeBench","oneLine":"LongJudgeBench evaluates LLM-as-a-judge performance on long-form outputs across six datasets covering pointwise, pairwise, and listwise protocols, with bilingual tasks and multiple prompt variants. It measures agreement with human judgments using accuracy, Spearman, and Kendall's tau.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01629","pdf":"https://arxiv.org/pdf/2606.01629","project":null,"code":"https://github.com/cjj826/LongJudgeBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.01629"},"evidence":{"snippet":"In this work, we introduce LongJudgeBench, a comprehensive benchmark for evaluating LLM judges on long-form outputs across diverse real-world scenarios and judging protocols.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01629"},"ranking":{},"description":"LongJudgeBench evaluates LLM-as-a-judge performance on long-form outputs across six datasets covering pointwise, pairwise, and listwise protocols, with bilingual tasks and multiple prompt variants. It measures agreement with human judgments using accuracy, Spearman, and Kendall's tau.","whyItMatters":"Existing meta-evaluation benchmarks focus on short-form outputs, leaving a gap for long-form evaluation. This benchmark provides a standardized way to assess judge reliability across diverse scenarios, helping practitioners select or develop judges for long-form tasks where current models show instability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9c4ac4a6122fbef7436d7a2b26ecff815871da43423d65c19bf5b4396ded1e84"},"motivation":"As large language models (LLMs) are increasingly used for long-form generation, reliably evaluating long-form outputs has become a critical challenge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01629","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LongJudgeBench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/cjj826/LongJudgeBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_longmedbench_c02feb5e","familyId":"bmf_b4036f59b4ef","name":"LongMedBench","oneLine":"A benchmark for long-horizon clinical decision-making using EHR data from MIMIC-IV, comprising 335 patients with multi-session interactions and three evaluation suites: fact-based QA, temporal reasoning, and long-horizon decision-making.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-07-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.09322","pdf":"https://arxiv.org/pdf/2607.09322","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.09322"},"evidence":{"snippet":"In this work, we introduce LongMedBench, a real-world EHR-based benchmark for long-horizon clinical decision-making.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.09322"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark for long-horizon clinical decision-making using EHR data from MIMIC-IV, comprising 335 patients with multi-session interactions and three evaluation suites: fact-based QA, temporal reasoning, and long-horizon decision-making.","whyItMatters":"Current medical agent evaluations emphasize short-context tasks, while real clinical care requires aggregating evidence over extended periods. This benchmark addresses the need for realistic long-horizon assessment, but the paper does not specify if the benchmark is publicly available for reuse or ongoing submission.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"12cf5e96de684d21fbe481b3e835415b102135ad2f514bdca70765f663539dc9"},"motivation":"In this work, we introduce LongMedBench, a real-world EHR-based benchmark for long-horizon clinical decision-making.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.09322","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_longrca-bench_4c5e770a","familyId":"bmf_a00231fd543d","name":"LongRCA Bench","oneLine":"LongRCA Bench is a benchmark for diagnosing responsible roles and root causes in long-horizon agent failures. It comprises 1,140 failed trajectories across five domains, with human labels for the responsible role and the earliest decisive root-cause step. Evaluation focuses on responsible-role accuracy and exact root-step accuracy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15242","pdf":"https://arxiv.org/pdf/2608.15242","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.15242"},"evidence":{"snippet":"We introduce LongRCA Bench, comprising 1,140 failed trajectories across five domains without injected errors.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":15,"hfDailySubmittedAt":"2026-08-25T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15242"},"ranking":{"30d":{"score":50,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":50,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"LongRCA Bench is a benchmark for diagnosing responsible roles and root causes in long-horizon agent failures. It comprises 1,140 failed trajectories across five domains, with human labels for the responsible role and the earliest decisive root-cause step. Evaluation focuses on responsible-role accuracy and exact root-step accuracy.","whyItMatters":"Long-horizon agent failures are difficult to debug, and outcome-level metrics obscure where errors occur. LongRCA Bench provides a standardized testbed for failure attribution, enabling comparison of methods that localize root causes and assign responsibility, which is crucial for improving agent reliability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"18a8bafb2f851f51d696b408dd44815dc1f03e4f22efa0eeb56cff0438e9cde5"},"motivation":"When a long-horizon agent execution fails, outcome-level evaluation reveals the unsuccessful result but not where the decisive error entered the trajectory.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15242","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LongRCA Bench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.15242","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_5d774bb2ffa891da","familyId":"catalog_family_5d774bb2ffa891da","name":"LongText-Bench","oneLine":"LongText-Bench evaluates text-to-image models on their ability to accurately render long text passages within generated images. It includes English (EN) and Chinese (ZH) subsets to assess multilingual text rendering capabilities.","description":"LongText-Bench evaluates text-to-image models on their ability to accurately render long text passages within generated images. It includes English (EN) and Chinese (ZH) subsets to assess multilingual text rendering capabilities.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Image-Generation","Language","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/longtext-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5d774bb2ffa891da"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/longtext-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"longtext-bench","url":"https://llm-stats.com/benchmarks/longtext-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image-generation","language","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_9eb6481a78fe9dbc","familyId":"catalog_family_9eb6481a78fe9dbc","name":"LongVideoBench","oneLine":"LongVideoBench is a question-answering benchmark featuring video-language interleaved inputs up to an hour long. It includes 3,763 varying-length web-collected videos with subtitles across diverse themes and 6,678 human-annotated multiple-choice questions in 17 fine-grained categories for comprehensive evaluation of long-term multimodal understanding.","description":"LongVideoBench is a question-answering benchmark featuring video-language interleaved inputs up to an hour long. It includes 3,763 varying-length web-collected videos with subtitles across diverse themes and 6,678 human-annotated multiple-choice questions in 17 fine-grained categories for comprehensive evaluation of long-term multimodal understanding.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/longvideobench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9eb6481a78fe9dbc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/longvideobench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"longvideobench","url":"https://llm-stats.com/benchmarks/longvideobench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","multimodal","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_longvqubench_aca1b92b","familyId":"bmf_324133a6cb63","name":"LongVQUBench","oneLine":"LongVQUBench evaluates long-term video quality understanding with 1200+ videos and 1500 questions across three levels of perceptual reasoning complexity.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.01086","pdf":"https://arxiv.org/pdf/2607.01086","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.01086"},"evidence":{"snippet":"To address these limitations, we present LongVQUBench, a comprehensive benchmark for long-term video quality understanding.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01086"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"LongVQUBench evaluates long-term video quality understanding with 1200+ videos and 1500 questions across three levels of perceptual reasoning complexity.","whyItMatters":"Fills a gap in video quality benchmarks by focusing on temporal continuity and cumulative degradation, with hierarchical evaluation levels for systematic assessment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ae62241396a6b7b64ede5b38df18a685936f72b9367d6c6ba0c8ce7aaeec0893"},"motivation":"The evaluation of long-term video quality understanding remains an open challenge for large vision-language models (LVLMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"European Conference on Computer Vision 2026","evidence":"Accepted at European Conference on Computer Vision 2026","evidenceUrl":"https://arxiv.org/abs/2607.01086","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"European Conference on Computer Vision 2026","reviewStatus":"accepted","decisionRaw":"Accepted at European Conference on Computer Vision 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.01086","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at European Conference on Computer Vision 2026","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_longwebbench_6f8a779e","familyId":"bmf_c347b5e54e89","name":"LongWebBench","oneLine":"LongWebBench evaluates structural and functional webpage generation of long webpages. It includes 490 webpages for structural fidelity and 129 for functional interactions, using VLM-based metrics and a DOM-augmented agent pipeline.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.17727","pdf":"https://arxiv.org/pdf/2606.17727","project":null,"code":"https://github.com/zheny2751-dotcom/LongWebBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.17727"},"evidence":{"snippet":"We introduce LongWebBench, a benchmark for evaluating long-horizon webpage generation from both structural and functional perspectives.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.17727"},"ranking":{"90d":{"score":23,"rank":379,"coverage":0.55,"confidence":"Low"}},"description":"LongWebBench evaluates structural and functional webpage generation of long webpages. It includes 490 webpages for structural fidelity and 129 for functional interactions, using VLM-based metrics and a DOM-augmented agent pipeline.","whyItMatters":"Current webpage generation benchmarks focus on short static pages, missing long-horizon coherence and interactive functionality. LongWebBench provides a reusable evaluation to assess models on executable multi-step interactions, which is critical for real-world deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a89ae7f8902c620b14857b9724a4e4f640055ba7ab1d4124ceb2e6d174b62bef"},"motivation":"Recent vision-language models (VLMs) have shown promising progress in generating webpages from visual inputs, yet existing evaluations mainly focus on short, single-screen, and largely static webpages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.17727","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LongWebBench team","organizationType":"benchmark-organization","sourceUrl":"https://github.com/zheny2751-dotcom/LongWebBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_longwof-bench_2e69a90a","familyId":"bmf_c4a7ef102c20","name":"LongWoF-Bench","oneLine":"Evaluates verifiable long-workflow tasks across code generation, agent-environment synthesis, mathematical reasoning, and rule following, with machine-verifiable scoring.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Reasoning","Code generation"],"topics":["Agents","Code","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.23200","pdf":"https://arxiv.org/pdf/2608.23200","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To evaluate this setting, we introduce the Long-Workflow Benchmark (LongWoF-Bench), comprising 778 machine-verifiable tasks across code generation, agent-environment synthesis, mathematical reasoning, and rule following.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":7,"hfDailySubmittedAt":"2026-08-25T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23200"},"ranking":{"30d":{"score":46,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":48,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates verifiable long-workflow tasks across code generation, agent-environment synthesis, mathematical reasoning, and rule following, with machine-verifiable scoring.","whyItMatters":"Provides a reusable benchmark for studying experience reuse in long-horizon LLM workflows, offering a standardized testbed for approaches like EvoMap.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"feff9333cde6b3b219b56e80b607257b37fe4d09872241b2107b756e7838e373"},"motivation":"Large language models are increasingly expected to execute complex workflows whose success depends on maintaining interdependent constraints and producing artifacts that satisfy strict end-to-end verification.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is explicitly introduced with a clear task set and verification mechanism, meeting the criteria for a public reusable benchmark.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce the Long-Workflow Benchmark (LongWoF-Bench), comprising 778 machine-verifiable tasks across code generation, agent-environment synthesis, mathematical reasoning, and rule following."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.23200","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":45,"confidence":"Low","horizon":"7d","reason":"The benchmark addresses workflow execution, a growing area, but the niche focus on experience reuse may limit broader immediate attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Agents","Coding & Software Engineering","Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_looparena_97bc3500","familyId":"bmf_0b2777f3ca58","name":"LoopArena","oneLine":"Loop Engineering is emerging as a practice for organizing development work around coding agents.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.28281v1","pdf":"https://arxiv.org/pdf/2608.28281v1","project":null,"code":"https://github.com/AMAP-ML/LoopArena","data":null,"hfPaper":null},"evidence":{"snippet":"We introduce LoopArena, a benchmark for evaluating how well one model can guide a separate coding agent through a long-running task.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.28281"},"ranking":{"30d":{"score":19,"rank":160,"coverage":0.85,"confidence":"High"},"90d":{"score":23,"rank":308,"coverage":0.7,"confidence":"Medium"}},"motivation":"Loop Engineering is emerging as a practice for organizing development work around coding agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T07:23:28.296315Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.28281v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_loopsbench_86719709","familyId":"bmf_0c0fee9fe39e","name":"LoopsBench","oneLine":"LOOPSBENCH evaluates coding agents on long-horizon tasks structured as dependency DAGs with flow-aware test release and regression obligations, spanning 8 languages and 9 domains.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00267","pdf":"https://arxiv.org/pdf/2608.00267","project":"https://loopsbench.ai/","code":"https://github.com/microsoft/Loopsbench","data":null,"hfPaper":"https://huggingface.co/papers/2608.00267"},"evidence":{"snippet":"We introduce LOOPSBENCH, a long-horizon benchmark for loop engineering in coding agent evaluation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-reviewed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":null,"githubStars":26,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00267"},"ranking":{"90d":{"score":43,"rank":127,"coverage":0.7,"confidence":"Medium"}},"description":"LOOPSBENCH evaluates coding agents on long-horizon tasks structured as dependency DAGs with flow-aware test release and regression obligations, spanning 8 languages and 9 domains.","whyItMatters":"Provides a benchmark for loop engineering in sustained software development, assessing planning, implementation, and recovery over long horizons.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"05fe1d8410d59b6afe5cdae1ed2a975ac29afc278a5a0b431692ebd9672933aa"},"motivation":"Coding agent infrastructure is shifting from harness engineering toward loop engineering as coding agents are deployed for sustained long-horizon software development.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"source-reviewed","reviewedAt":"2026-08-19","sources":["https://github.com/microsoft/Loopsbench","https://loopsbench.ai/","https://arxiv.org/abs/2608.00267"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00267","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"releaseDates":{"firstPublicAt":"2026-07-03","paperV1At":"2026-07-31"},"publishers":[{"name":"Microsoft","organizationType":"company-research-lab","sourceUrl":"https://github.com/microsoft/Loopsbench","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_lorft_f3adff80","familyId":"bmf_96f828c73143","name":"LoRFT","oneLine":"LoRFT evaluates long-range vehicle trajectory reconstruction from fixed highway cameras, with 6,601 manually verified trajectories and map-aware evaluation metrics like ADE and FDE.","area":"Vision & 3D","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Automotive"],"capabilities":["Factuality"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19911","pdf":"https://arxiv.org/pdf/2607.19911","project":null,"code":"https://github.com/YvfanZhu/LoRFT","data":null,"hfPaper":"https://huggingface.co/papers/2607.19911"},"evidence":{"snippet":"We introduce LoRFT, to our knowledge the first open benchmark dedicated to long-range vehicle trajectory reconstruction from fixed highway cameras.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19911"},"ranking":{"90d":{"score":35,"rank":200,"coverage":0.55,"confidence":"Low"}},"description":"LoRFT evaluates long-range vehicle trajectory reconstruction from fixed highway cameras, with 6,601 manually verified trajectories and map-aware evaluation metrics like ADE and FDE.","whyItMatters":"Long-range trajectories are essential for traffic safety and autonomous driving. A dedicated benchmark supports progress in reconstructing distant trajectories despite perspective compression and scale decay.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"09e502e90de8860f86c9bb97f6c9adc17727033f88f5923db33b385db1e45f42"},"motivation":"Long-range vehicle trajectories provide important spatio-temporal evidence for traffic safety analysis, autonomous driving evaluation, and data-driven traffic management, yet continuously recovering them from fixed highway cameras remains difficult.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19911","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LoRFT project","organizationType":"academic-lab","sourceUrl":"https://github.com/YvfanZhu/LoRFT","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_losona_d944301c","familyId":"bmf_d46a125f03ce","name":"LoSoNA","oneLine":"LoSoNA evaluates LLM agents' ability to infer and adapt to local social norms in group chats, using curated transcripts and an elicitor turn to score norm-conforming responses.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.14600","pdf":"https://arxiv.org/pdf/2606.14600","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.14600"},"evidence":{"snippet":"We introduce LoSoNA, a benchmark for local social norm adaptation in multi-party chat.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":7,"hfDailySubmittedAt":"2026-06-15T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.14600"},"ranking":{"90d":{"score":48,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"LoSoNA evaluates LLM agents' ability to infer and adapt to local social norms in group chats, using curated transcripts and an elicitor turn to score norm-conforming responses.","whyItMatters":"LLMs in social contexts must recognize unstated norms; this benchmark probes a capability that existing social benchmarks overlook, important for believable agent behavior.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cb381a3598a64e0209c366ffd430bcf8d7a451b7b81f1ed3fa84735f3db7a0d6"},"motivation":"Online group chats are social spaces with local conversational norms that are rarely stated explicitly.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.14600","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_lost-in-speech-trilingual-spoken-hallucina_f3c90115","familyId":"bmf_adb8b355243f","name":"Lost in Speech","oneLine":"A multilingual spoken hallucination detection benchmark with 12,013 news samples across English, Russian, and Kazakh, including synthetic and real-world fake news items in text and audio.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-26","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.24707","pdf":"https://arxiv.org/pdf/2608.24707","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present the first multilingual spoken hallucination benchmark comprising 12,013 news samples across English, Russian, and Kazakh with controlled hallucinations of three types and three severity levels.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24707"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A multilingual spoken hallucination detection benchmark with 12,013 news samples across English, Russian, and Kazakh, including synthetic and real-world fake news items in text and audio.","whyItMatters":"Addresses the gap in spoken hallucination detection, especially for low-resource languages, and evaluates transfer from synthetic to real-world fake news.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"7296574b009f67d46eac73320bb119a08e8b5f77b862516c10c2cdebf5ce692a"},"motivation":"While text-based hallucination detection has been extensively studied, spoken hallucination detection remains largely unexplored, particularly for low-resource languages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:08:47.245811Z","model":"deepseek-v4-pro","decisionReason":"Explicit benchmark corpus with controlled construction, real-world extensions, defined evaluation tasks (transcript vs audio), and results across multiple models.","canonicalNameSource":"paper_title","canonicalNameEvidence":"Lost in Speech: Trilingual Spoken Hallucination Detection Across Audio and Transcripts"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24707","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":40,"confidence":"Low","horizon":"7d","reason":"Niche topic but relevance to multilingual NLP may attract moderate interest; no public artifacts to drive visibility."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_34ee7686c8ea6671","familyId":"catalog_family_34ee7686c8ea6671","name":"LSAT","oneLine":"LSAT (Law School Admission Test) benchmark evaluating complex reasoning capabilities across three challenging tasks: analytical reasoning, logical reasoning, and reading comprehension. The LSAT measures skills considered essential for success in law school including critical thinking, reading comprehension of complex texts, and analysis of arguments.","description":"LSAT (Law School Admission Test) benchmark evaluating complex reasoning capabilities across three challenging tasks: analytical reasoning, logical reasoning, and reading comprehension. The LSAT measures skills considered essential for success in law school including critical thinking, reading comprehension of complex texts, and analysis of arguments.","area":"Language & Knowledge","applicationDomains":["Law & Government"],"primaryDomain":"Law & Government","industrySectors":[],"capabilities":[],"topics":["Legal","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/lsat","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_34ee7686c8ea6671"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/lsat"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"lsat","url":"https://llm-stats.com/benchmarks/lsat","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["legal","reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_lu-500_07127332","familyId":"bmf_71cc3bc652c4","name":"LU-500","oneLine":"LU-500 evaluates concept unlearning for logos with nearly 10,000 pairs, including explicit and implicit contextual tracks, and a multi-grained protocol measuring local removal and global preservation.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.24101","pdf":"https://arxiv.org/pdf/2607.24101","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.24101"},"evidence":{"snippet":"We introduce LU-500, a logo-unlearning benchmark built from Fortune Global 500 companies to study this localized and semantically entangled setting.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24101"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LU-500 evaluates concept unlearning for logos with nearly 10,000 pairs, including explicit and implicit contextual tracks, and a multi-grained protocol measuring local removal and global preservation.","whyItMatters":"Provides a specialized benchmark for a challenging unlearning scenario, enabling comparison of methods on localized and entangled visual concepts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"09a62db5a44d8e9be375ff396828b0ac449a5190dd8f42bb871865b7a7e59c04"},"motivation":"Concept unlearning is increasingly used to limit the reproduction of protected or unsafe visual concepts in text-to-image models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24101","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_lunar_024583f6","familyId":"bmf_9738b6bf3ae3","name":"LUNAR","oneLine":"LUNAR is a benchmark for evaluating LLM personalization from longitudinal app interaction logs across domains like clothing, food, housing, and mobility. It uses a synthetic data pipeline and evaluates 19 LLMs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.05246","pdf":"https://arxiv.org/pdf/2608.05246","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.05246"},"evidence":{"snippet":"To address this gap, we introduce LUNAR, the first benchmark for evaluating how LLMs personalize responses from longitudinal app interaction histories across universal daily-life domains, including clothing, food, housing, and mobility.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.05246"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"LUNAR is a benchmark for evaluating LLM personalization from longitudinal app interaction logs across domains like clothing, food, housing, and mobility. It uses a synthetic data pipeline and evaluates 19 LLMs.","whyItMatters":"LUNAR could address the need for evaluating cross-domain personalization from behavioral logs, but without public artifacts or clear scoring details, its utility is unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d4c52954918577de25a9629f41e917e2db0ebeadc7a853b7d3953571e3f18ec8"},"motivation":"Existing personalized LLM benchmarks primarily rely on textual personas or isolated behavioral signals, providing limited evaluation of cross-domain behavioral personalization, where responses must be grounded in heterogeneous daily-life activities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.05246","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_b029d6417276393b","familyId":"catalog_family_b029d6417276393b","name":"LVBench","oneLine":"LVBench is an extreme long video understanding benchmark designed to evaluate multimodal models on videos up to two hours in duration. It contains 6 major categories and 21 subcategories, with videos averaging five times longer than existing datasets. The benchmark addresses applications requiring comprehension of extremely long videos.","description":"LVBench is an extreme long video understanding benchmark designed to evaluate multimodal models on videos up to two hours in duration. It contains 6 major categories and 21 subcategories, with videos averaging five times longer than existing datasets. The benchmark addresses applications requiring comprehension of extremely long videos.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Long Context","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.8","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b029d6417276393b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/lvbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/lvbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"lvBench","url":"https://benchlm.ai/benchmarks/lvbench","paperUrl":"https://qwen.ai/blog?id=qwen3.8","year":"2026","fullName":"LVBench","format":"Long-video understanding score","tasks":"Long-form video question answering","successorKey":null},{"catalog":"llm-stats","sourceId":"lvbench","url":"https://llm-stats.com/benchmarks/lvbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","long context","multimodal","vision"],"catalogModelCount":26,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_l-tzcross-a-cross-lingual-page-level-bench_83560f19","familyId":"bmf_4882b79ad11f","name":"LëtzCross","oneLine":"Cross-lingual page-level retrieval benchmark over Luxembourgish PDF documents with queries in four languages.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-22","firstSeenAt":"2026-08-25","recognitionConfidence":0.45,"links":{"report":"http://arxiv.org/abs/2608.21714v1","pdf":"https://arxiv.org/pdf/2608.21714v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce LëtzCross, a benchmark for cross-lingual page-level retrieval over Luxembourgish PDF documents, with document pages indexed as images and queries provided in English, French, German, and Luxembourgish.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.21714"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Cross-lingual page-level retrieval benchmark over Luxembourgish PDF documents with queries in four languages.","whyItMatters":"Addresses low-resource and visually rich document retrieval, testing OCR-based and image-based retrievers.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"13bff14fd02ffa7f960d58468c6948a7f26afabd773dd6b5e6e021d08d4392b1"},"motivation":"Recent page-image retrievers such as ColPali have improved retrieval over visually rich documents, yet little is known about how they behave in cross-lingual, low-resource settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"No artifact evidence or public reuse path is provided, so the benchmark cannot be verified or reused.","canonicalNameSource":"paper_title","canonicalNameEvidence":"LëtzCross: A Cross-Lingual Page-Level Benchmark for Multimodal Retrieval over Luxembourgish Documents"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.21714v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":20,"confidence":"Low","horizon":"7d","reason":"The niche language focus and absence of released data or code lower expected attention."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Multimodal Perception","Search & Retrieval"],"domainScope":"general"},{"id":"bm_m-3-isr-a-multi-modal-multi-view-benchmark_dec2d75f","familyId":"bmf_812a8d67f9b0","name":"M$^3$ISR","oneLine":"Evaluates 3D/4D Gaussian splatting methods on 25 synthetic scenes with five tracks covering synthesis, streaming, and compression.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-25","recognitionConfidence":0.6,"links":{"report":"http://arxiv.org/abs/2608.22465v1","pdf":"https://arxiv.org/pdf/2608.22465v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce M$^3$ISR, a controlled synthetic benchmark for 3D and 4D Gaussian Splatting (3DGS/4DGS).","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22465"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates 3D/4D Gaussian splatting methods on 25 synthetic scenes with five tracks covering synthesis, streaming, and compression.","whyItMatters":"Controlled geometry and dense annotations enable systematic study of Gaussian-based FVV reconstruction and compression.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"ac2991508c4d72f0f4e8e6d2412faaf52540fe8c36f38dbc18840cfb9ea22fc3"},"motivation":"High-fidelity free-viewpoint video (FVV) and interactive rendering increasingly rely on explicit Gaussian representations, yet practical deployment remains constrained by representation size, dynamic updates, and computational cost.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is well-defined with multiple tracks and will be made available, supporting reproducible evaluation.","canonicalNameSource":"paper_title","canonicalNameEvidence":"M$^3$ISR: A Multi-Modal Multi-View Benchmark for 3D/4D Gaussian Splatting and Feedforward Compression"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22465v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":30,"confidence":"Medium","horizon":"7d","reason":"Gaussian splatting is a fast-moving area, but this synthetic benchmark may attract only niche interest due to its specialized focus."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_m3-duplexbench_afd47e79","familyId":"bmf_8b667f9a2394","name":"M3-DuplexBench","oneLine":"M3-DuplexBench evaluates full-duplex spoken dialogue models in multi-turn, multilingual (English and Japanese), multidomain settings, with multiple dialogue context settings and turn-taking analysis.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.29125","pdf":"https://arxiv.org/pdf/2607.29125","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.29125"},"evidence":{"snippet":"We propose M3-DuplexBench, a multi-turn, multilingual, multidomain benchmark for FDSDSs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.29125"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"M3-DuplexBench evaluates full-duplex spoken dialogue models in multi-turn, multilingual (English and Japanese), multidomain settings, with multiple dialogue context settings and turn-taking analysis.","whyItMatters":"Addresses the lack of fair multi-turn comparisons in full-duplex dialogue systems, enabling analysis across languages, domains, and context settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"db782de0dec032683a528a2956e516d9940a7a9d3ac98815983d456e9789d0d4"},"motivation":"Full-duplex spoken dialogue systems (FDSDSs) can listen while speaking, enabling natural behaviors such as smooth turn-taking, backchannel handling, and user barge-in handling.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.29125","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ma-proofbench_ec795126","familyId":"bmf_2dea3b94ab49","name":"MA-ProofBench","oneLine":"MA-ProofBench evaluates LLMs on theorem proving in mathematical analysis using 200 Lean 4 formalized problems split into undergraduate and Ph.D. levels. It covers six core topics and uses Pass@8 with formal verification via Lean server.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.13782","pdf":"https://arxiv.org/pdf/2606.13782","project":null,"code":"https://github.com/OpenBMB/MA-ProofBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.13782"},"evidence":{"snippet":"To address this gap, we introduce MA-ProofBench, to the best of our knowledge, the first formal theorem-proving benchmark dedicated to Mathematical Analysis.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13782"},"ranking":{"90d":{"score":30,"rank":243,"coverage":0.7,"confidence":"Medium"}},"description":"MA-ProofBench evaluates LLMs on theorem proving in mathematical analysis using 200 Lean 4 formalized problems split into undergraduate and Ph.D. levels. It covers six core topics and uses Pass@8 with formal verification via Lean server.","whyItMatters":"Formal theorem proving benchmarks typically lack coverage of advanced analysis. MA-ProofBench fills this gap, providing a stable evaluation for tracking progress in this difficult domain.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"89011505fc7da03fb120410278f5d7b8b383714e7b448165af92d4a4f07f5858"},"motivation":"Large Language Models (LLMs) have made notable progress in automated theorem proving, yet existing formal benchmarks remain limited in both mathematical coverage and difficulty.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13782","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"OpenBMB","organizationType":"community","sourceUrl":"https://github.com/OpenBMB/MA-ProofBench","role":"benchmark-publisher"}],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_mac-bench_4b2a0d04","familyId":"bmf_b6d1a8bb6911","name":"MAC-Bench","oneLine":"MAC-Bench evaluates procedural compliance of multi-agent systems under social-engineering pressure, measuring compliance-weighted success rate and Machiavellian gap.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07805","pdf":"https://arxiv.org/pdf/2606.07805","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07805"},"evidence":{"snippet":"To address this blind spot, we introduce MAC-Bench, a dynamic, adversarial benchmark designed to evaluate the procedural alignment of multi-agent systems under realistic pressure.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07805"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MAC-Bench evaluates procedural compliance of multi-agent systems under social-engineering pressure, measuring compliance-weighted success rate and Machiavellian gap.","whyItMatters":"Addresses 'Goodhart's Law' in agent alignment, offering metrics that trade off task success and compliance, but lacks a public evaluation path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d906f7ed3db57a5d8bcea62925481117965252c7b0d11952341df0bb82e972c7"},"motivation":"The rapid evolution of Large Language Models (LLMs) from passive assistants to autonomous, execution-capable agents has introduced critical operational risks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07805","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_macagentbench_782bfad8","familyId":"bmf_bd6b325f56e2","name":"MacAgentBench","oneLine":"MacAgentBench evaluates computer use agents on macOS desktop automation, with 676 tasks across 25 applications, including GUI and CLI interactions. It uses deterministic rule-based evaluation and fine-grained multi-checkpoint scoring to assess sub-goal completion.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22557","pdf":"https://arxiv.org/pdf/2606.22557","project":null,"code":"https://github.com/JetAstra/MacAgentBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.22557"},"evidence":{"snippet":"We present MacAgentBench, a comprehensive macOS agent benchmark comprising 676 tasks across 25 applications, with nearly 60% involving both GUI and CLI interaction.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":49,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22557"},"ranking":{"90d":{"score":44,"rank":115,"coverage":0.7,"confidence":"Medium"}},"description":"MacAgentBench evaluates computer use agents on macOS desktop automation, with 676 tasks across 25 applications, including GUI and CLI interactions. It uses deterministic rule-based evaluation and fine-grained multi-checkpoint scoring to assess sub-goal completion.","whyItMatters":"MacAgentBench addresses the need for benchmarks that capture framework-augmented agent capabilities and partial progress on long-horizon, multi-application tasks, providing a more granular comparison for real-world desktop automation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"826975e9c350fe0c6796ec56a396a5d639dfbc794ec5598e314bad1d4492384e"},"motivation":"Computer use agents (CUAs) have advanced rapidly in desktop automation, and a growing number of users deploy CUAs such as OpenClaw on Mac Mini for always-on automation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22557","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"JetAstra","organizationType":"community","sourceUrl":"https://github.com/JetAstra/MacAgentBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_macarena_a8690f15","familyId":"bmf_ff7f4b05b738","name":"MacArena","oneLine":"MacArena benchmarks computer-use agents on macOS with 421 manually verified tasks spanning 50 applications, running on Apple Virtualization framework on Apple Silicon.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06560","pdf":"https://arxiv.org/pdf/2606.06560","project":null,"code":"https://github.com/MacPaw/MacArena","data":null,"hfPaper":"https://huggingface.co/papers/2606.06560"},"evidence":{"snippet":"We introduce MacArena, a benchmark of 421 manually verified tasks spanning 50 applications that combines a curated port of OSWorld tasks, content sourced from macOSWorld, and 49 new macOS-native tasks, all running on Apple's native Virtualization framework on Apple Silicon.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":4,"hfDailySubmittedAt":null,"githubStars":10,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06560"},"ranking":{"90d":{"score":37,"rank":169,"coverage":0.7,"confidence":"Medium"}},"description":"MacArena benchmarks computer-use agents on macOS with 421 manually verified tasks spanning 50 applications, running on Apple Virtualization framework on Apple Silicon.","whyItMatters":"MacOS GUI challenges are underrepresented in current benchmarks. MacArena provides a harder and more diverse environment, revealing that performance on existing benchmarks may not generalize across platforms, aiding development of robust GUI agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c0a19fd0cd7a0e984a5aa7384b421bd1ef362b6ed51bb68f6bf9a022b9ece0dd"},"motivation":"Computer-use agents (CUAs) operate graphical user interfaces (GUIs) through vision and control primitives, and their capabilities have advanced rapidly, driven in part by standardized online evaluation benchmarks such as OSWorld, which serve both as evaluation tools and as training environments for reinforcement learning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"Second Workshop on Agents in the Wild: Safety, Security, and Beyond (AIWILD) at ICML 2026","evidence":"Accepted to the Second Workshop on Agents in the Wild: Safety, Security, and Beyond (AIWILD) at ICML 2026","evidenceUrl":"https://arxiv.org/abs/2606.06560","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"Second Workshop on Agents in the Wild: Safety, Security, and Beyond (AIWILD) at ICML 2026","reviewStatus":"accepted","decisionRaw":"Accepted to the Second Workshop on Agents in the Wild: Safety, Security, and Beyond (AIWILD) at ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.06560","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to the Second Workshop on Agents in the Wild: Safety, Security, and Beyond (AIWILD) at ICML 2026","level":"author-claim"}]}],"publishers":[{"name":"MacPaw","organizationType":"company-research-lab","sourceUrl":"https://github.com/MacPaw/MacArena","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_madb_64f075cb","familyId":"bmf_24baaf7d2a98","name":"MADB","oneLine":"MADB is a large-scale dataset and benchmark for music aesthetic assessment, comprising 9,999 tracks annotated by 30 trained annotators across 10 perceptual dimensions and an overall score, with textual comments. It includes a unified evaluation framework over multiple pretrained models.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SD"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.06929","pdf":"https://arxiv.org/pdf/2607.06929","project":null,"code":"https://github.com/knownree/madb","data":null,"hfPaper":"https://huggingface.co/papers/2607.06929"},"evidence":{"snippet":"We introduce MADB, a large-scale dataset and benchmark comprising 9,999 tracks annotated by 30 trained annotators.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06929"},"ranking":{"90d":{"score":33,"rank":218,"coverage":0.55,"confidence":"Low"}},"description":"MADB is a large-scale dataset and benchmark for music aesthetic assessment, comprising 9,999 tracks annotated by 30 trained annotators across 10 perceptual dimensions and an overall score, with textual comments. It includes a unified evaluation framework over multiple pretrained models.","whyItMatters":"Provides structured aesthetic annotations for a previously underexplored area, enabling measurement of model-human gaps in music understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9f80e509c552345f5b19ac57e8de74e1261ad68cc4d2df8aeb181a3b0024b086"},"motivation":"Music aesthetic assessment is a challenging yet underexplored problem, requiring models to capture fine-grained, multi-dimensional human perceptual judgments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06929","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Knownree","organizationType":"academic-lab","sourceUrl":"https://github.com/knownree/madb","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_madbench_d9b9325a","familyId":"bmf_14d0ddd00f54","name":"MADBench","oneLine":"MADBench is a benchmark for modality-aware audio deepfake detection, treating speech and environmental audio as distinct components. It evaluates detectors across independently manipulated forgery sources.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.SD"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.09593","pdf":"https://arxiv.org/pdf/2608.09593","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09593"},"evidence":{"snippet":"We introduce MADBench, the first benchmark that treats speech and environmental audio as distinct acoustic components, enabling component-aware evaluation of audio deepfake detection across independently manipulated forgery sources.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09593"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MADBench is a benchmark for modality-aware audio deepfake detection, treating speech and environmental audio as distinct components. It evaluates detectors across independently manipulated forgery sources.","whyItMatters":"It addresses the gap in audio deepfake detection by distinguishing speech and background audio, which have different generative mechanisms and artifact profiles, enabling component-aware evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"496597daaf8b11a6e0c4362f80da2c471a192314db625b5d7c4bb8a0b92f83ee"},"motivation":"Recent advances in speech synthesis and audio generation have made high-fidelity acoustic forgery low-cost and difficult to attribute, enabling a realistic attack scenario in which speech and background audio are independently manipulated over otherwise authentic video.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09593","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_madi-bench_da5b9728","familyId":"bmf_1f7a76f278ff","name":"MaDI-Bench","oneLine":"MaDI-Bench evaluates end-to-end data integration pipelines across schema matching, value normalization, entity matching, and conflict resolution.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.DB"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.30371","pdf":"https://arxiv.org/pdf/2606.30371","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.30371"},"evidence":{"snippet":"This paper fills this gap by introducing the Mannheim Data Integration Benchmark (MaDI-Bench), the first benchmark for the end-to-end integration of relational tables covering all steps of the integration process.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.30371"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MaDI-Bench evaluates end-to-end data integration pipelines across schema matching, value normalization, entity matching, and conflict resolution.","whyItMatters":"Data integration tasks are often evaluated piecemeal; an end-to-end benchmark could support comparison of holistic pipelines.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"366f7d1acdf55fabedacaf4f7c76759de4264be29cc1ecaed331cbeabb2d14af"},"motivation":"Data integration combines heterogeneous data sets into a single, coherent representation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.30371","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_mafiascope_6f84f9bf","familyId":"bmf_a84e60ac04f2","name":"MafiaScope","oneLine":"MafiaScope evaluates LLM agents in the social deduction game Mafia. It probes each agent's private beliefs after every public utterance, scoring them against ground truth without influencing the game. The testbed provides an open-source engine, interactive visualizer, recorded games, and counterfactual replay corpus.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.10645","pdf":"https://arxiv.org/pdf/2607.10645","project":"https://karpovilia.github.io/mafiascope/","code":"https://github.com/karpovilia/mafiascope","data":null,"hfPaper":"https://huggingface.co/papers/2607.10645"},"evidence":{"snippet":"We present MafiaScope, an open testbed that turns the social deduction game Mafia into a measurement instrument for machine Theory of Mind.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.10645"},"ranking":{"90d":{"score":23,"rank":372,"coverage":0.55,"confidence":"Low"}},"description":"MafiaScope evaluates LLM agents in the social deduction game Mafia. It probes each agent's private beliefs after every public utterance, scoring them against ground truth without influencing the game. The testbed provides an open-source engine, interactive visualizer, recorded games, and counterfactual replay corpus.","whyItMatters":"Machine Theory of Mind is difficult to measure from observable behavior alone. MafiaScope separates incorrect belief formation from incorrect action under correct beliefs, a distinction invisible in dialogue or outcome data, enabling more precise evaluation of social reasoning in LLM agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e824aa18d8728d8ce133e2b350308995e6708abafda4ae2fe48e5e64a53b88ea"},"motivation":"An LLM agent's public behaviour reveals little about its social reasoning: an agent that votes correctly may be guessing, and an agent that lies well leaves no trace of what it actually believes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.10645","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MafiaScope Team","organizationType":"academic-lab","sourceUrl":"https://github.com/karpovilia/mafiascope","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_mag_3c2b6829","familyId":"bmf_f8685538bda2","name":"MAG","oneLine":"MAG is a benchmark that unifies task execution and guide writing into a single multimodal action and guide task, with grounding over screenshots. It includes a harness for annotation, training, evaluation, and joint metrics.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.10079","pdf":"https://arxiv.org/pdf/2607.10079","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.10079"},"evidence":{"snippet":"In this work we introduce MAG, the first benchmark that unifies task execution and guide writing into a single Multimodal Action and Guide task, with two grounding schemes over screenshots: Set-of-Mark element selection and raw pixel coordinates.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.10079"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MAG is a benchmark that unifies task execution and guide writing into a single multimodal action and guide task, with grounding over screenshots. It includes a harness for annotation, training, evaluation, and joint metrics.","whyItMatters":"The evaluation gap is that prior benchmarks separate web agent actions and guide text generation, and often rely on textual DOM rather than screenshots. MAG provides a unified evaluation for multimodal understanding and generation in live environments, which could support development of more capable web agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f07973bc7589c7ceb062e24adbecc848593c2bb8b8b2d41f9e3336833c1996ce"},"motivation":"Digital Adoption Platforms (DAPs) are embedded overlays widely used on web systems to guide users through operations inside a page, helping them get started with unfamiliar interfaces quickly.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.10079","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_maliciousskillbench_08ed3dc5","familyId":"bmf_0285cbc8ba0d","name":"MaliciousSkillBench","oneLine":"Benchmark for detecting malicious agent skills, comprising 9,740 skills (7,505 malicious, 2,235 benign) and evaluating detectors under random, structural-disjoint, and source-disjoint splits.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-21","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.19901","pdf":"https://arxiv.org/pdf/2608.19901","project":"https://protectskills.github.io/MaliciousSkillBench/","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present MaliciousSkillBench, a comprehensive benchmark for malicious Agent Skill detection.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.19901"},"ranking":{"30d":{"score":40,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Benchmark for detecting malicious agent skills, comprising 9,740 skills (7,505 malicious, 2,235 benign) and evaluating detectors under random, structural-disjoint, and source-disjoint splits.","whyItMatters":"Addresses the need for robust malicious skill detection by providing a consolidated cross-source dataset and evaluation protocols that measure both detection and false-positive rates.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"26519b25e53beac4cec813dd1535878febdf57535c076da579e4cd40a5d10762"},"motivation":"Agent Skills extend LLM agents with reusable instruction packages that may also include scripts, resources, and service configuration.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The paper explicitly introduces the benchmark with a stable scoring contract and the official project page signals public availability of resources.","canonicalNameSource":"paper_title","canonicalNameEvidence":"MaliciousSkillBench: A Comprehensive Benchmark for Malicious Agent Skill Detection"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.19901","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-21T04:30:09.569171Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses security risks in agent skills, a timely topic, and includes comprehensive evaluation results likely to draw interest from the AI safety community."},"evaluationMode":"public_reusable","publishers":[{"name":"ProtectSkills","organizationType":"benchmark-organization","sourceUrl":"https://protectskills.github.io/MaliciousSkillBench/","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_malskillbench_80086691","familyId":"bmf_f88e2b4dd1af","name":"MalSkillBench","oneLine":"MalSkillBench is a runtime-verified benchmark of malicious agent skills, with 3,944 malicious skills labeled along a taxonomy, measuring detection tool effectiveness.","area":"Language & Knowledge","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Logistics"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07131","pdf":"https://arxiv.org/pdf/2606.07131","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07131"},"evidence":{"snippet":"We present MalSkillBench, the first runtime-verified benchmark of malicious agent skills: 3,944 malicious skills labeled along a three-dimensional taxonomy of 108 cells.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07131"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MalSkillBench is a runtime-verified benchmark of malicious agent skills, with 3,944 malicious skills labeled along a taxonomy, measuring detection tool effectiveness.","whyItMatters":"Evaluates detection tools for hybrid code-prompt skills, potentially informing supply chain security, but lacks public artifacts for reuse.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"349e093176a4625ee481c4416df6e4cdc46986bb6731ea513196b84549699a6b"},"motivation":"AI coding agents such as Claude Code and Gemini CLI increasingly extend themselves with third-party skills: markdown packages bundling natural-language instructions, executable scripts, and tool permissions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07131","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_mamabench_7ce134aa","familyId":"bmf_db4a560447e2","name":"MamaBench","oneLine":"MamaBench evaluates LLM robustness in maternal and child health diagnosis using counterfactual clinical narratives, with the Bias Trap Rate (BTR) metric.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Robustness"],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-07-15","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.14385","pdf":"https://arxiv.org/pdf/2607.14385","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.14385"},"evidence":{"snippet":"We introduce MamaBench, the first counterfactual benchmark for maternal and paediatric AI: 434 expert-authored clinical narratives in 217 pairs across 371 pathologies, evaluated via the Bias Trap Rate (BTR), the conditional probability that a model fails the counterfactual given success on the base case.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14385"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MamaBench evaluates LLM robustness in maternal and child health diagnosis using counterfactual clinical narratives, with the Bias Trap Rate (BTR) metric.","whyItMatters":"It highlights the gap between base accuracy and robust accuracy in clinical AI, showing that models can fail on clinically similar cases requiring different interventions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"79b8aad9e670b441115c9dcf37c6f69a3a9c3e37fc478b338a5de9d518cc2a19"},"motivation":"Large language models achieve strong scores on medical benchmarks, yet these benchmarks evaluate each question in isolation, providing no measure of whether a system can distinguish clinically similar presentations requiring different interventions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14385","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"catalog_fd6a9290a3fa92d3","familyId":"catalog_family_fd6a9290a3fa92d3","name":"Management Consulting Tasks (Internal)","oneLine":"Management Consulting Tasks is an internal OpenAI evaluation of long-horizon professional knowledge work drawn from management-consulting workflows, scoring whether models produce correct, decision-ready analyses.","description":"Management Consulting Tasks is an internal OpenAI evaluation of long-horizon professional knowledge work drawn from management-consulting workflows, scoring whether models produce correct, decision-ready analyses.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Reasoning","Finance","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/management-consulting-tasks","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fd6a9290a3fa92d3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/management-consulting-tasks"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"management-consulting-tasks","url":"https://llm-stats.com/benchmarks/management-consulting-tasks","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","finance","agents"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_mandate-bench_a638082e","familyId":"bmf_692dbb1815c5","name":"Mandate Bench","oneLine":"Evaluates LLM agents' compliance with portfolio-rebalancing mandates through frozen market snapshots and pre-registered metrics.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/msavdert/mandate-bench","pdf":null,"project":"https://msavdert.github.io/mandate-bench/","code":"https://github.com/msavdert/mandate-bench","data":null,"hfPaper":null},"evidence":{"snippet":"mandate-bench Behavioral benchmarking of LLM agents on a fixed portfolio-rebalancing task: mandate compliance, decision consistency, reasoning-action agreement agents benchmark instruction-following llm reproducibility # Mandate Bench Behavioral benchmarking of LLM agents on a fixed portfolio-rebalancing task: mandate compliance, decision consistency, and reasoning-action agreement.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:msavdert/mandate-bench"},"ranking":{"30d":{"score":23,"rank":119,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":323,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates LLM agents' compliance with portfolio-rebalancing mandates through frozen market snapshots and pre-registered metrics.","whyItMatters":"Shifts agent evaluation from noisy profit metrics to behavioral consistency and rule adherence, offering a reproducible standard for financial agents.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"07600f11c57ab5436a033651399aabbfed3990969d927aab771360e6a25643a9"},"motivation":"mandate-bench Behavioral benchmarking of LLM agents on a fixed portfolio-rebalancing task: mandate compliance, decision consistency, reasoning-action agreement agents benchmark instruction-following llm reproducibility # Mandate Bench Behavioral benchmarking of LLM agents on a fixed portfolio-rebalancing task: mandate compliance, decision consistency, and reasoning-action agreement.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/msavdert/mandate-bench","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses a timely gap in AI trading agent evaluation with reproducible methodology and public leaderboard."},"evaluationMode":"public_reusable","publishers":[{"name":"Mandate Bench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/msavdert/mandate-bench","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_maniguard-a-benchmark-and-data-suite-for-s_fde8045a","familyId":"bmf_b025643cd9d4","name":"ManiGuard-Bench","oneLine":"ManiGuard-Bench evaluates 1,000 locked scenarios across six contact-rich household task families, with safety specifications checked by LTL$_f$-grounded automaton monitors and metrics separating safe success from engaged-and-safe behavior.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.17386","pdf":"https://arxiv.org/pdf/2608.17386","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce ManiGuard, a specification-grounded framework for evaluating and improving the safety of foundation-model manipulation, comprising the ManiGuard-Bench task suite and a paired safety-annotated trajectory-generation pipeline.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","named entity is a model, method, framework, or dataset used in evaluation"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17386"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"ManiGuard-Bench evaluates 1,000 locked scenarios across six contact-rich household task families, with safety specifications checked by LTL$_f$-grounded automaton monitors and metrics separating safe success from engaged-and-safe behavior.","whyItMatters":"ManiGuard-Bench fills a gap by evaluating robotic manipulation safety independently of task success, revealing that up to 21% of successful rollouts violate safety specifications and enabling safety-aware fine-tuning with 8,000 demonstrations.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"02fbaa3f9373140b328e5afcac18c9e19cbf4f86db63191c68c1377d6e3e92c9"},"motivation":"Foundation-model policies for robotic manipulation are advancing rapidly on task success, but rigorous evaluation of whether they succeed safely is still lacking.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally named in the abstract, defines repeatable locked tasks with fixed safety specifications and monitor-based evaluation, and provides a public path through released demonstrations and a pipeline.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce ManiGuard, a specification-grounded framework for evaluating and improving the safety of foundation-model manipulation, comprising the ManiGuard-Bench task suite and a paired safety-annotated trajectory-generation pipeline."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17386","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets safety in foundation-model manipulation with large-scale rollouts and hardware validation, likely attracting attention from robotics and AI safety communities."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_mapreason-osm_48426257","familyId":"bmf_f407dc71cb3b","name":"MapReason-OSM","oneLine":"MapReason-OSM evaluates vision-language models on graph-verifiable mobility decisions from self-rendered OpenStreetMap panels. It covers 12 tasks in routing, facility location, and visual disambiguation, with structured outputs scored against hidden oracles for validity, legality, optimality, and constraint satisfaction, plus cross-zoom consistency.","area":"Multimodal","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Logistics"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22597","pdf":"https://arxiv.org/pdf/2606.22597","project":null,"code":"https://github.com/Vi-Sri/mapreason-osm","data":null,"hfPaper":"https://huggingface.co/papers/2606.22597"},"evidence":{"snippet":"We present MapReason-OSM, a benchmark and evaluation harness for graph-verifiable mobility decisions on self-rendered OpenStreetMap panels.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22597"},"ranking":{"90d":{"score":23,"rank":377,"coverage":0.55,"confidence":"Low"}},"description":"MapReason-OSM evaluates vision-language models on graph-verifiable mobility decisions from self-rendered OpenStreetMap panels. It covers 12 tasks in routing, facility location, and visual disambiguation, with structured outputs scored against hidden oracles for validity, legality, optimality, and constraint satisfaction, plus cross-zoom consistency.","whyItMatters":"Existing map benchmarks often rely on free-text or multiple-choice answers that cannot be verified against road networks. This benchmark provides a reproducible, exact scoring contract for decision-making tasks, supporting comparison of VLM capabilities in logistics, delivery, and accessible navigation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"805fbe51bda3d2baa37217f3de7124299a9f0d3715ee546ff553f515c8261a57"},"motivation":"Vision-language models (VLMs) are increasingly used to read maps for logistics, delivery, and accessible navigation, where the output is an actionable decision (a route, a pin, a parking choice) that must respect the road network.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22597","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Vi-Sri","organizationType":"community","sourceUrl":"https://github.com/Vi-Sri/mapreason-osm","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_5b294ac811a2d33e","familyId":"catalog_family_5b294ac811a2d33e","name":"Market-Bench","oneLine":"A quantitative-trading implementation benchmark that asks models to build backtesters under market-book liquidity and execution-delay constraints, then compares their outputs with a verifier.","description":"A quantitative-trading implementation benchmark that asks models to build backtesters under market-book liquidity and execution-delay constraints, then compares their outputs with a verifier.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2512.12264","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5b294ac811a2d33e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/marketbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"marketBench","url":"https://benchlm.ai/benchmarks/marketbench","paperUrl":"https://arxiv.org/abs/2512.12264","year":"2025","fullName":"Market-Bench","format":"Backtester implementation scored by mean absolute error","tasks":"3 quantitative-trading strategies","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_masdrift_b9a896af","familyId":"bmf_1a3bba8a4a37","name":"MasDrift","oneLine":"Evaluates multi-agent systems on 600 productivity tasks measuring task completion and unauthorized action rate across different coordination architectures.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07556","pdf":"https://arxiv.org/pdf/2608.07556","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07556"},"evidence":{"snippet":"We introduce MasDrift, a benchmark of 600 benign productivity tasks across eight domains.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07556"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates multi-agent systems on 600 productivity tasks measuring task completion and unauthorized action rate across different coordination architectures.","whyItMatters":"Makes authorization preservation a measurable property of MAS design, exposing trade-offs between centralized and decentralized coordination for safe delegation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"b2296dd68558a7875bc749a9c2d60434f87a6d534cdd13b1835ce0be673e4547"},"motivation":"Multi-agent systems (MAS) decompose long-horizon tasks across supervisors and subagents, but delegated goals do not necessarily carry their original authorization boundaries.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07556","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MasDrift Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.07556","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_mash-bench_aa5b1c8c","familyId":"bmf_58c350f668c2","name":"MASH-Bench","oneLine":"Evaluates ML models on cross-source mass-shooting risk classification using 6,968 incidents with leave-one-dataset-out evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-25","recognitionConfidence":0.85,"links":{"report":"http://arxiv.org/abs/2608.22460v1","pdf":"https://arxiv.org/pdf/2608.22460v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce MASH-Bench, a harmonized benchmark of 6,968 incidents from four U.S.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22460"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates ML models on cross-source mass-shooting risk classification using 6,968 incidents with leave-one-dataset-out evaluation.","whyItMatters":"Cross-source generalization is critical for risk classification, and this benchmark isolates the role of feature completeness and label prevalence.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"958e475c4634e9145341febd4209733c505a102a87b5a17f7f3873a1cb5ffde8"},"motivation":"Public mass-shooting databases differ substantially in coverage, feature availability, and reporting practices, creating challenges for machine-learning models that must generalize across data sources.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark provides a harmonized dataset and controlled evaluation protocol, and the paper serves as a public path for reuse.","canonicalNameSource":"paper_title","canonicalNameEvidence":"MASH-Bench: Diagnosing Cross-Source Failure in Mass-Shooting Risk Classification"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22460v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":20,"confidence":"Medium","horizon":"7d","reason":"The topic is highly specialized and may have limited appeal beyond researchers in public safety and domain adaptation."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_48bf9b4f142ab45c","familyId":"catalog_family_48bf9b4f142ab45c","name":"MASK","oneLine":"MASK is a collection of 1000 questions measuring whether models faithfully report their beliefs when pressured to lie. It operationalizes deception as the rate at which the model lies, i.e., knowingly making false statements intended to be received as true. Lower dishonesty rates indicate better honesty.","description":"MASK is a collection of 1000 questions measuring whether models faithfully report their beliefs when pressured to lie. It operationalizes deception as the rate at which the model lies, i.e., knowingly making false statements intended to be received as true. Lower dishonesty rates indicate better honesty.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mask","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_48bf9b4f142ab45c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mask"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mask","url":"https://llm-stats.com/benchmarks/mask","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","safety"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_mat-pref_c4647f88","familyId":"bmf_fe32cc058aa2","name":"Mat-Pref","oneLine":"Mat-Pref evaluates compositional reasoning in inorganic materials via 10,837 ionic-substitution questions across 11 structure families, with splits for in-distribution performance, held-out families, and cross-property transfer.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.21830","pdf":"https://arxiv.org/pdf/2606.21830","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.21830"},"evidence":{"snippet":"We introduce Mat-Pref, a benchmark of 10,837 ionic-substitution questions across 11 inorganic structure families, grounded in density functional theory calculations from the Materials Project, with three evaluation splits that isolate in-distribution performance, generalization to entirely held-out structure families, and cross-property transfer: applying band-gap reasoning to hosts seen during training only through formation-energy supervision.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.21830"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Mat-Pref evaluates compositional reasoning in inorganic materials via 10,837 ionic-substitution questions across 11 structure families, with splits for in-distribution performance, held-out families, and cross-property transfer.","whyItMatters":"It isolates generalization types (structural transfer, property transfer, memorization) in scientific reasoning, helping to identify when RLVR improves reasoning over memorization.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3bac324eb215eab8e185a43d43fa52900eea9cd33d52800fdb374ce972795b22"},"motivation":"Reinforcement learning from verifiable rewards (RLVR) has driven rapid progress in mathematical and code reasoning, but when extended to science, existing benchmarks do not decompose what generalizes: do gains reflect structural transfer, property transfer, or memorization?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML AI4Physics 2026 Workshop","evidence":"10 pages, 4 figures, Accepted at ICML AI4Physics 2026 Workshop","evidenceUrl":"https://arxiv.org/abs/2606.21830","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML AI4Physics 2026 Workshop","reviewStatus":"accepted","decisionRaw":"10 pages, 4 figures, Accepted at ICML AI4Physics 2026 Workshop","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.21830","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"10 pages, 4 figures, Accepted at ICML AI4Physics 2026 Workshop","level":"author-claim"}]}],"publishers":[{"name":"Mat-Pref team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.21830","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_58a6d6801ae771e6","familyId":"catalog_family_58a6d6801ae771e6","name":"MATH","oneLine":"MATH dataset contains 12,500 challenging competition mathematics problems from AMC 10, AMC 12, AIME, and other mathematics competitions. Each problem includes full step-by-step solutions and spans multiple difficulty levels (1-5) across seven mathematical subjects including Prealgebra, Algebra, Number Theory, Counting and Probability, Geometry, Intermediate Algebra, and Precalculus.","description":"MATH dataset contains 12,500 challenging competition mathematics problems from AMC 10, AMC 12, AIME, and other mathematics competitions. Each problem includes full step-by-step solutions and spans multiple difficulty levels (1-5) across seven mathematical subjects including Prealgebra, Algebra, Number Theory, Counting and Probability, Geometry, Intermediate Algebra, and Precalculus.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_58a6d6801ae771e6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mathbenchmark"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/math"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mathBenchmark","url":"https://benchlm.ai/benchmarks/mathbenchmark","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"MATH","format":"Exact match","tasks":"Competition math problems","successorKey":null},{"catalog":"llm-stats","sourceId":"math","url":"https://llm-stats.com/benchmarks/math","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":71,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_9eb10ec16be22efb","familyId":"catalog_family_9eb10ec16be22efb","name":"MATH (CoT)","oneLine":"MATH dataset contains 12,500 challenging competition mathematics problems from AMC 10, AMC 12, AIME, and other mathematics competitions. Each problem includes full step-by-step solutions and spans multiple difficulty levels (1-5) across seven mathematical subjects. This variant uses Chain-of-Thought prompting to encourage step-by-step reasoning.","description":"MATH dataset contains 12,500 challenging competition mathematics problems from AMC 10, AMC 12, AIME, and other mathematics competitions. Each problem includes full step-by-step solutions and spans multiple difficulty levels (1-5) across seven mathematical subjects. This variant uses Chain-of-Thought prompting to encourage step-by-step reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/math-(cot)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9eb10ec16be22efb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/math-(cot)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"math-(cot)","url":"https://llm-stats.com/benchmarks/math-(cot)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"lib_math_500","familyId":"family_math","name":"MATH-500","oneLine":"Established benchmark variant · Mathematical Reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Mathematical Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2103.03874","pdf":null,"project":"https://github.com/hendrycks/math","code":"https://github.com/hendrycks/math","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_math_500"},"ranking":{},"recordType":"variant","aliases":["MATH 500"],"sourceAttribution":[{"role":"parent-dataset-paper","url":"https://arxiv.org/abs/2103.03874"}],"adoptionRefs":["openai-gpt5","deepseek-v3"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"deepseek-v3","url":"https://github.com/deepseek-ai/DeepSeek-V3","provider":"DeepSeek","model":"DeepSeek-V3"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"variantOfExternal":"MATH","capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"math500","url":"https://benchlm.ai/benchmarks/math-500","paperUrl":"https://arxiv.org/abs/2103.03874","year":"2021","fullName":"MATH-500 Problem Set","format":"Free-form mathematical answers","tasks":"500 problems","successorKey":null},{"catalog":"benchlm","sourceId":"valsMath500","url":"https://benchlm.ai/benchmarks/valsmath500","paperUrl":"https://www.vals.ai/benchmarks/math500","year":"2026","fullName":"Vals MATH 500","format":"Accuracy score","tasks":"MATH 500 academic math problems","successorKey":null},{"catalog":"llm-stats","sourceId":"math-500","url":"https://llm-stats.com/benchmarks/math-500","datasetSlug":"math-500","versionCount":1,"subsetCount":1,"rowCount":500,"updatedAt":"2026-06-05T18:58:16.851840+00:00","community":true}],"catalogCategories":["math","external","reasoning"],"catalogModelCount":32,"catalogStarCount":0},{"id":"bm_math-vision-diagrams_40787688","familyId":"bmf_54b9a73db7dc","name":"Math-Vision Diagrams","oneLine":"Math-Vision Diagrams evaluates LLMs on mathematical diagram generation from text, covering both text-to-code and text-to-image paradigms, with a curated subset of 2,920 competition problems and multiple metrics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.08964","pdf":"https://arxiv.org/pdf/2608.08964","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.08964"},"evidence":{"snippet":"We introduce Math-Vision Diagrams, the first benchmark specifically designed to evaluate LLMs on mathematical diagram generation, and the first to assess text-to-code and text-to-image generation paradigms together in a single unified setting, agnostic of the underlying coding lan- guage or model type.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08964"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Math-Vision Diagrams evaluates LLMs on mathematical diagram generation from text, covering both text-to-code and text-to-image paradigms, with a curated subset of 2,920 competition problems and multiple metrics.","whyItMatters":"It fills the gap in standardized evaluation of math diagram generation, enabling comparison across paradigms and models, and provides a comprehensive benchmark for this emerging capability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a93b713c407a583ff127d3b2a3e591b713c9f6ce56c2dd934bb71782bfbf916b"},"motivation":"The generation of mathematically precise diagrams from tex- tual prompts has emerged as a critical yet underexplored capability of Large Language Models (LLMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICANN 2026","evidence":"Accepted at ICANN 2026","evidenceUrl":"https://arxiv.org/abs/2608.08964","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICANN 2026","reviewStatus":"accepted","decisionRaw":"Accepted at ICANN 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.08964","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at ICANN 2026","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception","Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_mathadv_85c33564","familyId":"bmf_9e215fa673d4","name":"MathAdv","oneLine":"Evaluates theorem proving and auxiliary tasks including multiple-choice, fill-in-the-blank, and reformulation robustness across 13 mathematical domains.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-08-26","firstSeenAt":"2026-08-27","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25449","pdf":"https://arxiv.org/pdf/2608.25449","project":null,"code":"https://github.com/margotyjx/MathAdv.git","data":null,"hfPaper":null},"evidence":{"snippet":"We introduce MathAdv, a diagnostic benchmark spanning 13 domains across undergraduate- and graduate-level mathematics.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25449"},"ranking":{"30d":{"score":34,"rank":65,"coverage":0.55,"confidence":"Low"},"90d":{"score":33,"rank":213,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates theorem proving and auxiliary tasks including multiple-choice, fill-in-the-blank, and reformulation robustness across 13 mathematical domains.","whyItMatters":"Offers component-wise diagnosis of formal reasoning beyond aggregate proof accuracy, exposing formalization and robustness gaps.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"4f9d6ce04f6ec7b8e469791e82f2bb6f6da1ef56912619d78c5d2c3c55b77b52"},"motivation":"Formal theorem proving enables machine-verifiable evaluation of mathematical reasoning, yet existing benchmarks often emphasize aggregate proof accuracy, concentrate on a narrow range of mathematics, and provide limited evidence of robustness to equivalent reformulations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:12:36.511123Z","model":"deepseek-v4-pro","decisionReason":"MathAdv is explicitly named and described as a benchmark, and the abstract reports dataset and evaluation scripts available on GitHub, though the link is unverified.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce MathAdv, a diagnostic benchmark spanning 13 domains across undergraduate- and graduate-level mathematics."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25449","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":50,"confidence":"Low","horizon":"7d","reason":"Focused formal theorem proving benchmark with diagnostic tasks, but limited to Lean 4 and linked repository status unverified."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_f1a7872c89a6bca4","familyId":"catalog_family_f1a7872c89a6bca4","name":"MathArena Apex","oneLine":"MathArena Apex is a challenging math contest benchmark featuring the most difficult mathematical problems designed to test advanced reasoning and problem-solving abilities of AI models. It focuses on olympiad-level mathematics and complex multi-step mathematical reasoning.","description":"MathArena Apex is a challenging math contest benchmark featuring the most difficult mathematical problems designed to test advanced reasoning and problem-solving abilities of AI models. It focuses on olympiad-level mathematics and complex multi-step mathematical reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/matharena-apex","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f1a7872c89a6bca4"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/matharena-apex"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"matharena-apex","url":"https://llm-stats.com/benchmarks/matharena-apex","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_92c8d52c1cda7e67","familyId":"catalog_family_92c8d52c1cda7e67","name":"MathVerse","oneLine":"MathVerse evaluates multimodal mathematical reasoning, testing whether models genuinely interpret visual math diagrams rather than relying on text.","description":"MathVerse evaluates multimodal mathematical reasoning, testing whether models genuinely interpret visual math diagrams rather than relying on text.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mathverse","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_92c8d52c1cda7e67"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mathverse"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mathverse","url":"https://llm-stats.com/benchmarks/mathverse","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","multimodal","reasoning","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_e552b2299f18d7c5","familyId":"catalog_family_e552b2299f18d7c5","name":"MathVerse-Mini","oneLine":"MathVerse-Mini is a subset of the MathVerse benchmark for evaluating math reasoning capabilities in vision-language models.","description":"MathVerse-Mini is a subset of the MathVerse benchmark for evaluating math reasoning capabilities in vision-language models.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mathverse-mini","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e552b2299f18d7c5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mathverse-mini"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mathverse-mini","url":"https://llm-stats.com/benchmarks/mathverse-mini","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","multimodal","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_f7c236f33a18c30e","familyId":"catalog_family_f7c236f33a18c30e","name":"MathVision","oneLine":"MATH-Vision is a dataset designed to measure multimodal mathematical reasoning capabilities. It focuses on evaluating how well models can solve mathematical problems that require both visual understanding and mathematical reasoning, bridging the gap between visual and mathematical domains.","description":"MATH-Vision is a dataset designed to measure multimodal mathematical reasoning capabilities. It focuses on evaluating how well models can solve mathematical problems that require both visual understanding and mathematical reasoning, bridging the gap between visual and mathematical domains.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Math","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f7c236f33a18c30e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mathvision"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mathvision"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mathVision","url":"https://benchlm.ai/benchmarks/mathvision","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"MathVision","format":"Image + math reasoning","tasks":"Visually grounded math problems","successorKey":null},{"catalog":"llm-stats","sourceId":"mathvision","url":"https://llm-stats.com/benchmarks/mathvision","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","math","multimodal","vision"],"catalogModelCount":34,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_0d097e7f564943e8","familyId":"catalog_family_0d097e7f564943e8","name":"MathVision w/ Python","oneLine":"A tool-augmented MathVision variant that permits Python during visual mathematics reasoning.","description":"A tool-augmented MathVision variant that permits Python during visual mathematics reasoning.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0d097e7f564943e8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mathvisionpython"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mathVisionPython","url":"https://benchlm.ai/benchmarks/mathvisionpython","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"MathVision with Python","format":"Image and mathematics reasoning with tools","tasks":"Visual mathematics problems with Python","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_84bea39f6a34f3cf","familyId":"catalog_family_84bea39f6a34f3cf","name":"MathVista","oneLine":"MathVista evaluates mathematical reasoning of foundation models in visual contexts. It consists of 6,141 examples derived from 28 existing multimodal datasets and 3 newly created datasets (IQTest, FunctionQA, and PaperQA), combining challenges from diverse mathematical and visual tasks to assess models' ability to understand complex figures and perform rigorous reasoning.","description":"MathVista evaluates mathematical reasoning of foundation models in visual contexts. It consists of 6,141 examples derived from 28 existing multimodal datasets and 3 newly created datasets (IQTest, FunctionQA, and PaperQA), combining challenges from diverse mathematical and visual tasks to assess models' ability to understand complex figures and perform rigorous reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mathvista","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_84bea39f6a34f3cf"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mathvista"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mathvista","url":"https://llm-stats.com/benchmarks/mathvista","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","multimodal","vision"],"catalogModelCount":39,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_e7b6a8fd37aa43a8","familyId":"catalog_family_e7b6a8fd37aa43a8","name":"MathVista-Mini","oneLine":"MathVista-Mini is a smaller version of the MathVista benchmark that evaluates mathematical reasoning in visual contexts. It consists of examples derived from multimodal datasets involving mathematics, combining challenges from diverse mathematical and visual tasks to assess foundation models' ability to solve problems requiring both visual understanding and mathematical reasoning.","description":"MathVista-Mini is a smaller version of the MathVista benchmark that evaluates mathematical reasoning in visual contexts. It consists of examples derived from multimodal datasets involving mathematics, combining challenges from diverse mathematical and visual tasks to assess foundation models' ability to solve problems requiring both visual understanding and mathematical reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mathvista-mini","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e7b6a8fd37aa43a8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mathvista-mini"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mathvista-mini","url":"https://llm-stats.com/benchmarks/mathvista-mini","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","multimodal","vision"],"catalogModelCount":25,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_matphasebench_87d9e805","familyId":"bmf_77353b9cc428","name":"MatPhaseBench","oneLine":"MatPhaseBench evaluates vision-language models on understanding materials phase diagrams, using 200 diagram-text pairs from 3681 papers. It targets complex scientific image understanding, with tasks requiring deep comprehension and open-ended responses, covering 189 material systems and 70 elements.","area":"Vision & 3D","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.02934","pdf":"https://arxiv.org/pdf/2607.02934","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.02934"},"evidence":{"snippet":"We introduce MatPhaseBench, a high-quality, high-reliability benchmark for complex scientific image understanding, focused on materials phase diagrams.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02934"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MatPhaseBench evaluates vision-language models on understanding materials phase diagrams, using 200 diagram-text pairs from 3681 papers. It targets complex scientific image understanding, with tasks requiring deep comprehension and open-ended responses, covering 189 material systems and 70 elements.","whyItMatters":"This benchmark addresses the gap in evaluating VLMs on logically complex scientific diagrams that require mechanistic reasoning. It measures capabilities beyond surface perception, helping assess practical value for AI-assisted materials science analysis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6f1bce36a4116ef185ebf2f826a847eb0d532740d3c957c3e61b53082e4ccd79"},"motivation":"Materials phase diagrams are a core knowledge representation in materials science, encoding temperature,composition, phase stability, and phase transformation pathways, with their full understanding requiring thermodynamic mechanism analysis and scientific reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02934","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_matreplace_a6f0342f","familyId":"bmf_b16c649ffad8","name":"MatReplace","oneLine":"A reference-free benchmark for material replacement in interior scenes, evaluating edits on local material correctness, global lighting harmony, outside preservation, and inside structure across three conditioning tracks.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.24107","pdf":"https://arxiv.org/pdf/2608.24107","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce MatReplace, a reference-free benchmark that evaluates edits along four verifiable dimensions: local material correctness, global lighting harmony, outside preservation, and inside structure.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24107"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A reference-free benchmark for material replacement in interior scenes, evaluating edits on local material correctness, global lighting harmony, outside preservation, and inside structure across three conditioning tracks.","whyItMatters":"Provides a standardized evaluation for a commercially relevant task where reference-based metrics are inadequate, enabling fair comparison of editors under different conditioning signals.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"b9792d55ccabf6cc1969069f9bc94a47e9518372dc7fd1c45907689ea7452f11"},"motivation":"Material replacement is a common interior-design operation: changing the material of a selected surface while preserving its geometry, surroundings, and illumination.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:08:47.245811Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with explicit evaluation protocol, three tracks, expert-validated metrics, and clear performance analysis.","canonicalNameSource":"paper_title","canonicalNameEvidence":"MatReplace: A Reference-Free, Conditioning-Aligned Benchmark for Material Replacement in Interior Scenes"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24107","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":45,"confidence":"Low","horizon":"7d","reason":"Topic has moderate breadth in computer vision and design, but lack of public artifacts limits immediate visibility."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_448a1c38fad37f5e","familyId":"catalog_family_448a1c38fad37f5e","name":"MAVERIX","oneLine":"MAVERIX (Multimodal Audio-Visual Evaluation Reasoning Index) evaluates multimodal models on tasks that demand tight integration of video and audio information. It features challenges like situational awareness and social sentiment analysis where the answer cannot be reliably determined from a single modality, rigorously testing joint audio-visual understanding.","description":"MAVERIX (Multimodal Audio-Visual Evaluation Reasoning Index) evaluates multimodal models on tasks that demand tight integration of video and audio information. It features challenges like situational awareness and social sentiment analysis where the answer cannot be reliably determined from a single modality, rigorously testing joint audio-visual understanding.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Audio","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/maverix","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_448a1c38fad37f5e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/maverix"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"maverix","url":"https://llm-stats.com/benchmarks/maverix","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","audio","video","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_b3114cc4196c2e25","familyId":"catalog_family_b3114cc4196c2e25","name":"MAXIFE","oneLine":"MAXIFE is a multilingual benchmark evaluating LLMs on instruction following and execution across multiple languages and cultural contexts.","description":"MAXIFE is a multilingual benchmark evaluating LLMs on instruction following and execution across multiple languages and cultural contexts.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multilingual","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b3114cc4196c2e25"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/maxife"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/maxife"}],"catalogSources":[{"catalog":"benchlm","sourceId":"maxife","url":"https://benchlm.ai/benchmarks/maxife","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"MAXIFE","format":"Cross-lingual benchmark","tasks":"Multilingual instruction following","successorKey":null},{"catalog":"llm-stats","sourceId":"maxife","url":"https://llm-stats.com/benchmarks/maxife","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multilingual","general"],"catalogModelCount":11,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_mba-bench_3f910264","familyId":"bmf_50ad9d899e34","name":"MBA-Bench","oneLine":"Evaluates multimodal business ideation agents across six domains using 30K samples, with MLLM-as-a-Judge scoring over six business-oriented criteria.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":0.75,"links":{"report":"https://arxiv.org/abs/2608.11616","pdf":"https://arxiv.org/pdf/2608.11616","project":"https://hchoi256.github.io/projects/mba/","code":"https://github.com/hchoi256/MBA","data":null,"hfPaper":null},"evidence":{"snippet":"We thus introduce MBA-Bench, the first multimodal benchmark for training and evaluating business ideation agents, comprising 30K samples across six domains, each domain characterized by distinct visual cues not fully conveyed by text alone.","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":6,"hfDailySubmittedAt":"2026-08-13T00:00:00.000Z","githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11616"},"ranking":{"30d":{"score":33,"rank":69,"coverage":0.85,"confidence":"High"},"90d":{"score":34,"rank":207,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates multimodal business ideation agents across six domains using 30K samples, with MLLM-as-a-Judge scoring over six business-oriented criteria.","whyItMatters":"Provides the first multimodal benchmark for business ideation, enabling comparison of agents that ground ideas in diverse real-world visual contexts beyond text-only approaches.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"6dfbdf595b580b665d8fd4a6c86743ce2eaaf75e758d63b30164bba1816c02ab"},"motivation":"Agentic systems powered by large language models (LLMs) have opened new opportunities for business ideation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with 30K samples, explicit evaluation criteria, public dataset, code, and project page enabling repeatable scoring and submission.","canonicalNameSource":"abstract","canonicalNameEvidence":"We thus introduce MBA-Bench, the first multimodal benchmark for training and evaluating business ideation agents"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11616","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"Novel multimodal business ideation task scope, large dataset, released code and demo, and arXiv paper with project page support moderate community interest."},"evaluationMode":"score_submission","publishers":[{"name":"KAIST AI","organizationType":"academic-lab","sourceUrl":"https://hchoi256.github.io/projects/mba/","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mbench_6b4d0471","familyId":"bmf_5facad24d801","name":"MBench","oneLine":"MBench is a benchmark for memory capability of video world models, decomposing into entity, environment, and causal consistency with 12 sub-dimensions. Includes code, dataset, and leaderboard.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.00793","pdf":"https://arxiv.org/pdf/2606.00793","project":"https://peanutup.github.io/MBench-project/","code":"https://github.com/study-overflow/MBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.00793"},"evidence":{"snippet":"To address this gap, we present \\textbf{MBench}, a comprehensive benchmark dedicated to quantifying and evaluating the memory capability of video world models.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":11,"hfDailySubmittedAt":null,"githubStars":117,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00793"},"ranking":{},"description":"MBench is a benchmark for memory capability of video world models, decomposing into entity, environment, and causal consistency with 12 sub-dimensions. Includes code, dataset, and leaderboard.","whyItMatters":"Fills the gap in evaluating long-term state retention in video world models, providing a standardized benchmark to advance the field.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4dd57fe4a8e34045c4559d47bf92d0d04c3fa4113f3cf8c0d5a3442c3930d75b"},"motivation":"Recent advancements in video-based world models have demonstrated an unprecedented ability to synthesize high-fidelity visual sequences.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00793","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_0c2ab8aed657e87b","familyId":"catalog_family_0c2ab8aed657e87b","name":"MBPP","oneLine":"MBPP+ is an enhanced version of MBPP (Mostly Basic Python Problems) with significantly more test cases (35x) for more rigorous evaluation. MBPP is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers, covering programming fundamentals and standard library functionality.","description":"MBPP+ is an enhanced version of MBPP (Mostly Basic Python Problems) with significantly more test cases (35x) for more rigorous evaluation. MBPP is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers, covering programming fundamentals and standard library functionality.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mbpp","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0c2ab8aed657e87b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mbpp"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mbpp+"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mbpp","url":"https://llm-stats.com/benchmarks/mbpp","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false},{"catalog":"llm-stats","sourceId":"mbpp+","url":"https://llm-stats.com/benchmarks/mbpp+","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":33,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_f2ed6b6b0d44aa63","familyId":"catalog_family_f2ed6b6b0d44aa63","name":"MBPP ++ base version","oneLine":"MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases covering programming fundamentals and standard library functionality. This is an enhanced version with additional test cases.","description":"MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases covering programming fundamentals and standard library functionality. This is an enhanced version with additional test cases.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mbpp-++-base-version","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f2ed6b6b0d44aa63"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mbpp-++-base-version"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mbpp-++-base-version","url":"https://llm-stats.com/benchmarks/mbpp-++-base-version","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e4fa459970018e88","familyId":"catalog_family_e4fa459970018e88","name":"MBPP EvalPlus","oneLine":"MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. EvalPlus extends MBPP with significantly more test cases (35x) for more rigorous evaluation of LLM-synthesized code, providing high-quality and precise evaluation.","description":"MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. EvalPlus extends MBPP with significantly more test cases (35x) for more rigorous evaluation of LLM-synthesized code, providing high-quality and precise evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mbpp-evalplus","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e4fa459970018e88"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mbpp-evalplus"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mbpp-evalplus","url":"https://llm-stats.com/benchmarks/mbpp-evalplus","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_8ab302d3bad1721c","familyId":"catalog_family_8ab302d3bad1721c","name":"MBPP EvalPlus (base)","oneLine":"MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. EvalPlus extends MBPP with significantly more test cases (35x) for more rigorous evaluation of LLM-synthesized code, providing high-quality and precise evaluation.","description":"MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. EvalPlus extends MBPP with significantly more test cases (35x) for more rigorous evaluation of LLM-synthesized code, providing high-quality and precise evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mbpp-evalplus-(base)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8ab302d3bad1721c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mbpp-evalplus-(base)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mbpp-evalplus-(base)","url":"https://llm-stats.com/benchmarks/mbpp-evalplus-(base)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_e583ae35322cb0f6","familyId":"catalog_family_e583ae35322cb0f6","name":"MBPP pass@1","oneLine":"MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases. This variant uses pass@1 evaluation metric measuring the percentage of problems solved correctly on the first attempt.","description":"MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases. This variant uses pass@1 evaluation metric measuring the percentage of problems solved correctly on the first attempt.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mbpp-pass@1","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e583ae35322cb0f6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mbpp-pass@1"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mbpp-pass@1","url":"https://llm-stats.com/benchmarks/mbpp-pass@1","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_3b988625ad11d391","familyId":"catalog_family_3b988625ad11d391","name":"MBPP Plus","oneLine":"MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases covering programming fundamentals and standard library functionality. This is an enhanced version with additional test cases for more rigorous evaluation.","description":"MBPP (Mostly Basic Python Problems) is a benchmark of 974 crowd-sourced Python programming problems designed to be solvable by entry-level programmers. Each problem consists of a task description, code solution, and 3 automated test cases covering programming fundamentals and standard library functionality. This is an enhanced version with additional test cases for more rigorous evaluation.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mbpp-plus","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3b988625ad11d391"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mbpp-plus"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mbpp-plus","url":"https://llm-stats.com/benchmarks/mbpp-plus","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_mc-cxr-a-multi-context-chest-x-ray-benchma_1a149873","familyId":"bmf_bfa02518247e","name":"MC-CXR","oneLine":"MC-CXR is a benchmark of 240 chest X-ray cases (2,522 instances) for evaluating context-induced disruption in vision-language models. It pairs reliable and misleading context across text and prior imaging, with visual overlays. Defines three task families and two paired metrics: switch-to-wrong rate and context-aligned error rate.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-26","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.24118","pdf":"https://arxiv.org/pdf/2608.24118","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce Multi-Context Chest X-ray (MC-CXR), a benchmark of 240 cases expanded into 2,522 instances that isolates context-induced disruption through paired perturbation.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24118"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"today":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MC-CXR is a benchmark of 240 chest X-ray cases (2,522 instances) for evaluating context-induced disruption in vision-language models. It pairs reliable and misleading context across text and prior imaging, with visual overlays. Defines three task families and two paired metrics: switch-to-wrong rate and context-aligned error rate.","whyItMatters":"VLMs in clinical pipelines may be disrupted by plausible but misleading context, reducing diagnostic accuracy. MC-CXR isolates this disruption with paired perturbation, revealing high switch rates and text-visual asymmetry, supporting safer deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"913f7206eab8533f99cbdacbc971d32924b7f9e0cc3c2729f60371d20bbdd9fb"},"motivation":"Vision-language models (VLMs) are increasingly used in clinical pipelines where a chest X-ray is interpreted alongside retrieved reports, preliminary notes, or prior imaging.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The dataset is available on PhysioNet, providing a public reuse path, and the benchmark defines clear paired metrics.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce Multi-Context Chest X-ray (MC-CXR), a benchmark"},"publication":{"status":"acceptance_claimed","venue":"Findings of EMNLP 2026","evidence":"15 pages, 3 figures, 4 tables. Accepted to Findings of EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2608.24118","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-26T06:08:22.244942Z"},"venueAttempts":[{"venueName":"Findings of EMNLP 2026","reviewStatus":"accepted","decisionRaw":"15 pages, 3 figures, 4 tables. Accepted to Findings of EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.24118","observedAt":"2026-08-26T06:08:22.244942Z","rawValue":"15 pages, 3 figures, 4 tables. Accepted to Findings of EMNLP 2026","level":"author-claim"}]}],"attentionForecast":{"score":70,"confidence":"Medium","horizon":"7d","reason":"Clinical AI relevance and dataset availability on PhysioNet likely attract medical imaging and VLM researchers."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_544e399fd1681104","familyId":"catalog_family_544e399fd1681104","name":"MCP Atlas","oneLine":"MCP Atlas is a benchmark for evaluating AI models on scaled tool use capabilities, measuring how well models can coordinate and utilize multiple tools across complex multi-step tasks.","description":"MCP Atlas is a benchmark for evaluating AI models on scaled tool use capabilities, measuring how well models can coordinate and utilize multiple tools across complex multi-step tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agentic","Reasoning","Agents","Code","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_544e399fd1681104"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mcpatlas"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mcp-atlas"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mcpAtlas","url":"https://benchlm.ai/benchmarks/mcpatlas","paperUrl":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","year":"2026","fullName":"MCP Atlas","format":"Interactive tool-calling evaluation","tasks":"Tool-integrated agent tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"mcp-atlas","url":"https://llm-stats.com/benchmarks/mcp-atlas","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","agents","code","tool calling"],"catalogModelCount":33,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_1e40c583106167df","familyId":"catalog_family_1e40c583106167df","name":"MCP Mark Verified","oneLine":"A human-verified edition of MCPMark for MCP tool use across Notion, GitHub, Filesystem, Postgres, and Playwright server environments.","description":"A human-verified edition of MCPMark for MCP tool use across Notion, GitHub, Filesystem, Postgres, and Playwright server environments.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://mcpmark.ai/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1e40c583106167df"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mcpmarkverified"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mcpMarkVerified","url":"https://benchlm.ai/benchmarks/mcpmarkverified","paperUrl":"https://mcpmark.ai/","year":"2026","fullName":"MCPMark-Verified","format":"Interactive MCP task completion","tasks":"MCP tool-use tasks across five server environments","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_cf947aca0afb2ab0","familyId":"catalog_family_cf947aca0afb2ab0","name":"MCP-Atlas claim coverage","oneLine":"Average coverage of required claims in answers produced during real-world MCP tool-use workflows.","description":"Average coverage of required claims in answers produced during real-world MCP tool-use workflows.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_cf947aca0afb2ab0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mcpatlasclaimcoverage"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mcpAtlasClaimCoverage","url":"https://benchlm.ai/benchmarks/mcpatlasclaimcoverage","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"MCP-Atlas mean claim coverage","format":"Mean claim coverage","tasks":"Production-like multi-server MCP workflows","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_99d3067c4c9f0c37","familyId":"catalog_family_99d3067c4c9f0c37","name":"MCP-Mark","oneLine":"MCP-Mark evaluates LLMs on their ability to use Model Context Protocol (MCP) tools effectively, testing tool discovery, selection, invocation, and result interpretation across diverse MCP server scenarios.","description":"MCP-Mark evaluates LLMs on their ability to use Model Context Protocol (MCP) tools effectively, testing tool discovery, selection, invocation, and result interpretation across diverse MCP server scenarios.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mcp-mark","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_99d3067c4c9f0c37"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mcp-mark"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mcp-mark","url":"https://llm-stats.com/benchmarks/mcp-mark","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","tool calling"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_mcp-persona_e4dd478c","familyId":"bmf_d0faf0a6b15e","name":"MCP-Persona","oneLine":"MCP-Persona evaluates LLM agents on real-world personalized MCP tools across social media, collaboration, email, and content management applications. It includes 173 tool-chain tasks, 139 unique tools, and 18 MCP servers, with a fully automated environment simulation pipeline for reproducible evaluation.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02470","pdf":"https://arxiv.org/pdf/2606.02470","project":null,"code":"https://github.com/wwh0411/MCP-Persona","data":null,"hfPaper":"https://huggingface.co/papers/2606.02470"},"evidence":{"snippet":"To bridge this critical gap, we introduce MCP-Persona, the first benchmark specifically designed for evaluating agent performance on real-world, personalized MCP tools.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":16,"hfDailySubmittedAt":null,"githubStars":7,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02470"},"ranking":{},"description":"MCP-Persona evaluates LLM agents on real-world personalized MCP tools across social media, collaboration, email, and content management applications. It includes 173 tool-chain tasks, 139 unique tools, and 18 MCP servers, with a fully automated environment simulation pipeline for reproducible evaluation.","whyItMatters":"Existing benchmarks overlook personalized MCP tool use, which is critical for practical agent deployment. MCP-Persona provides a sandboxed, reproducible environment to test agents on realistic personal tasks, helping identify limitations and guiding development of more capable agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4521a2d9c11d1d10cc5db2c38447b4f940b0a15a163d1df64fa97ed9d8ab8741"},"motivation":"The Model Context Protocol (MCP) has emerged as a transformative standard for connecting large language models (LLMs) with external data sources and tools, and has been rapidly adopted across personal applications and development platforms.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02470","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MCP-Persona team","organizationType":"academic-lab","sourceUrl":"https://github.com/wwh0411/MCP-Persona","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_de1ff364516d521c","familyId":"catalog_family_de1ff364516d521c","name":"MCP-Tasks","oneLine":"A Model Context Protocol task benchmark used in Qwen's launch tables to measure practical execution over MCP-style tools and integrations.","description":"A Model Context Protocol task benchmark used in Qwen's launch tables to measure practical execution over MCP-style tools and integrations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_de1ff364516d521c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mcptasks"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mcpTasks","url":"https://benchlm.ai/benchmarks/mcptasks","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"MCP-Tasks","format":"Interactive tool-use evaluation","tasks":"MCP-integrated tool tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_945d8dca6fae8ba8","familyId":"catalog_family_945d8dca6fae8ba8","name":"MCP-Universe","oneLine":"MCP-Universe evaluates LLMs on complex multi-step agentic tasks using Model Context Protocol (MCP) tools across diverse interactive environments, testing planning, tool orchestration, and task completion.","description":"MCP-Universe evaluates LLMs on complex multi-step agentic tasks using Model Context Protocol (MCP) tools across diverse interactive environments, testing planning, tool orchestration, and task completion.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mcp-universe","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_945d8dca6fae8ba8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mcp-universe"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mcp-universe","url":"https://llm-stats.com/benchmarks/mcp-universe","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","tool calling"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_mcpevol-bench_3df86292","familyId":"bmf_dfd5c20c84be","name":"MCPEvol-Bench","oneLine":"The evaluation object is unclear from the provided information.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14642","pdf":"https://arxiv.org/pdf/2607.14642","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.14642"},"evidence":{"snippet":"To bridge this gap, we introduce \\textbf{MCPEvol-Bench}, a novel benchmark for evaluating the task-solving capabilities of LLM agents under dynamic toolset evolution.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14642"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"The evaluation object is unclear from the provided information.","whyItMatters":"The evaluation gap and practical decision value are unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"30cc1765ad8772c9749b1a33ed33fb2eeb23e606abae7c9278f663596aa577bd"},"motivation":"As Model Context Protocol (MCP) servers emerge as the core infrastructure for connecting LLMs with external tools, existing benchmarks leverage real-world MCP servers to evaluate LLM agents' tool-using capabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14642","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_mcr-bench_10494b73","familyId":"bmf_0ade0c7988a9","name":"MCR-Bench","oneLine":"MCR-Bench is a benchmark for evaluating reproducibility in mission-critical LLM tasks, measuring output consistency across heterogeneous hardware.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-19","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.21023","pdf":"https://arxiv.org/pdf/2606.21023","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.21023"},"evidence":{"snippet":"To evaluate our approach practically, we introduce MCR-Bench, a benchmark targeting reproducibility in mission-critical tasks.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.21023"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MCR-Bench is a benchmark for evaluating reproducibility in mission-critical LLM tasks, measuring output consistency across heterogeneous hardware.","whyItMatters":"LLM deployments in finance, medicine, and law require reproducible outputs. MCR-Bench addresses the lack of standardized evaluation for numerical instability in 16-bit inference, helping practitioners select methods that balance reproducibility and performance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d0e1eb36a5bf2c868962434555e77c8adc1ca4aef192bf057ce81e714bd3b6a9"},"motivation":"As Large Language Models (LLMs) deploy into mission-critical domains (e.g., finance, medicine, and law), output reproducibility has become a strict system requirement.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.21023","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_mcr-bench_7a6da290","familyId":"bmf_0ade0c7988a9","name":"MCR-Bench","oneLine":"In real-world software development, code review typically involves iterative interactions between developers and reviewers to improve software quality, making the process costly and time-consuming.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.27442","pdf":"https://arxiv.org/pdf/2608.27442","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.27442"},"evidence":{"snippet":"To bridge this gap, we introduce MCR-Bench, the first defect state-aware benchmark designed for realistic multi-round code review.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27442"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"In real-world software development, code review typically involves iterative interactions between developers and reviewers to improve software quality, making the process costly and time-consuming.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"acceptance_claimed","venue":"ISSTA 2026","evidence":"Accepted at ISSTA 2026","evidenceUrl":"https://arxiv.org/abs/2608.27442","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ISSTA 2026","reviewStatus":"accepted","decisionRaw":"Accepted at ISSTA 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.27442","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at ISSTA 2026","level":"author-claim"}]}],"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_post-training-vlms-for-video-mistake-detec_58e7640e","familyId":"bmf_5cfb2a67196a","name":"MD-VQA","oneLine":"Tests video models on detecting whether a step was executed correctly according to its description, for seen and unseen actions.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.28406","pdf":"https://arxiv.org/pdf/2608.28406","project":null,"code":"https://github.com/FedeSpu/mstk","data":null,"hfPaper":null},"evidence":{"snippet":"To reflect this, we introduce the Mistake Detection Video Question Answering (MD-VQA) protocol and accompanying benchmark.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.28406"},"ranking":{"30d":{"score":28,"rank":85,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":255,"coverage":0.55,"confidence":"Low"}},"description":"Tests video models on detecting whether a step was executed correctly according to its description, for seen and unseen actions.","whyItMatters":"Shifts mistake detection to open-set evaluation, requiring models to understand general mistake concepts rather than memorizing steps.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T04:51:30.440525Z","inputHash":"d6e4e78f3ee8fb905902e95f6bb79bf17d3720511698cdf643b67cb9964e7241"},"motivation":"Human mistakes are inevitable when following instructions, yet they can lead to severe consequences.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T04:51:30.440525Z","model":"deepseek-v4-pro","decisionReason":"Formally named MD-VQA protocol with code and benchmark released on GitHub.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce the Mistake Detection Video Question Answering (MD-VQA) protocol and accompanying benchmark."},"publication":{"status":"acceptance_claimed","venue":"BMVC 2026","evidence":"Accepted at BMVC 2026","evidenceUrl":"https://arxiv.org/abs/2608.28406","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T01:03:30.163531Z"},"venueAttempts":[{"venueName":"BMVC 2026","reviewStatus":"accepted","decisionRaw":"Accepted at BMVC 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.28406","observedAt":"2026-08-31T01:03:30.163531Z","rawValue":"Accepted at BMVC 2026","level":"author-claim"}]}],"attentionForecast":{"score":70,"confidence":"Medium","horizon":"7d","reason":"Combines a novel open-set protocol with video-language post-training, and explicit code release, in a field receiving growing attention."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mdarena_a3bdd691","familyId":"bmf_7ef0c7c56d53","name":"MDArena","oneLine":"MDArena evaluates coding agents on 50 containerized molecular dynamics tasks from active research projects, covering trajectory analysis, system preparation, free-energy protocols, and enhanced sampling, with strict pass-based scoring.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["physics.chem-ph"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02642","pdf":"https://arxiv.org/pdf/2608.02642","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.02642"},"evidence":{"snippet":"To address this issue, we introduce MDArena, a benchmark of 50 containerized tasks drawn from active biomolecular simulation projects, spanning 29 molecular systems and 14 broad research protocols, including trajectory analysis, complex system preparation, free-energy protocols, and enhanced sampling.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02642"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MDArena evaluates coding agents on 50 containerized molecular dynamics tasks from active research projects, covering trajectory analysis, system preparation, free-energy protocols, and enhanced sampling, with strict pass-based scoring.","whyItMatters":"Bridges the gap between supervised AI assistance and autonomous research reliability by providing a reproducible platform for tracking progress in automating complex scientific workflows.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"63b89ecbcbac162af4010b27a3e19f290c35d34aff1eae74b897e5d271103f0c"},"motivation":"Accelerating scientific discovery is among the most consequential applications of AI, and computational biomolecular simulation stands out as a particularly promising target within this broader effort.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02642","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_f9f1763b8efbc7c6","familyId":"catalog_family_f9f1763b8efbc7c6","name":"MeasureBench","oneLine":"MeasureBench evaluates multimodal models on visual measurement and quantitative perception tasks across both real and synthetic imagery, reported as the average over the two settings.","description":"MeasureBench evaluates multimodal models on visual measurement and quantitative perception tasks across both real and synthetic imagery, reported as the average over the two settings.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/measurebench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f9f1763b8efbc7c6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/measurebench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"measurebench","url":"https://llm-stats.com/benchmarks/measurebench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mecobench_fd729c4e","familyId":"bmf_a28f8f826055","name":"MECoBench","oneLine":"MECoBench is a multimodal embodied cooperation benchmark with an evaluation platform. It spans real-world tasks, two cooperation structures, and three collaboration modes, with code and dataset publicly available.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Agents","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.31966","pdf":"https://arxiv.org/pdf/2606.31966","project":"https://q-i-n-g.github.io/MECoBench-Website/","code":"https://github.com/q-i-n-g/MECoBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.31966"},"evidence":{"snippet":"To address this gap, we introduce MECoBench, a multimodal embodied cooperation benchmark with an evaluation platform spanning diverse real-world tasks, two cooperation structures, and three collaboration modes.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31966"},"ranking":{"90d":{"score":36,"rank":185,"coverage":0.55,"confidence":"Low"}},"description":"MECoBench is a multimodal embodied cooperation benchmark with an evaluation platform. It spans real-world tasks, two cooperation structures, and three collaboration modes, with code and dataset publicly available.","whyItMatters":"Systematically evaluates collaboration among multimodal embodied agents, a relatively unexplored area. Provides a testbed for understanding collaboration mechanisms and limits.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d96a96bde9c953799192146b51b052286595c4cecd1518fec4286fed54d9273e"},"motivation":"Recent multimodal large language models (MLLMs) have strong potential as embodied agents, but their ability to collaborate in visually grounded environments remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31966","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"q-i-n-g","organizationType":"academic-lab","sourceUrl":"https://github.com/q-i-n-g/MECoBench","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_medbench_c5b43204","familyId":"bmf_9629d422c4d1","name":"MedBench","oneLine":"MedBench v5 evaluates clinical multimodal models across 63 tasks with process-oriented metrics including stressors and hallucination tracking.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Agents","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24155","pdf":"https://arxiv.org/pdf/2606.24155","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.24155"},"evidence":{"snippet":"We introduce MedBench v5, a redesigned benchmark for clinical multimodal models (language, vision-language, and agent systems) that moves from static QA to dynamic, process-oriented evaluation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24155"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MedBench v5 evaluates clinical multimodal models across 63 tasks with process-oriented metrics including stressors and hallucination tracking.","whyItMatters":"Addresses gaps in process visibility and hallucination detection in medical AI evaluation, offering a unified framework for capability profiling and stress testing.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3f5ec9ae3c543561fed00836ed0e0126898a6bbee8aa121270d4d71de598a871"},"motivation":"Existing medical AI benchmarks lack process visibility, atomic skill evaluation, and integrated hallucination detection.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24155","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_2b9a45a0de6fd5b9","familyId":"catalog_family_2b9a45a0de6fd5b9","name":"MedChemBench (Internal)","oneLine":"MedChemBench is an internal OpenAI evaluation of medicinal-chemistry reasoning, testing whether models can support drug-discovery-relevant chemistry analysis and decision-making.","description":"MedChemBench is an internal OpenAI evaluation of medicinal-chemistry reasoning, testing whether models can support drug-discovery-relevant chemistry analysis and decision-making.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Reasoning","Science","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/medchembench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2b9a45a0de6fd5b9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/medchembench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"medchembench","url":"https://llm-stats.com/benchmarks/medchembench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","science","healthcare"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_medclawbench_bcd0f6e4","familyId":"bmf_56afb80b1bd4","name":"MedClawBench","oneLine":"MedClawBench evaluates long-horizon temporal reasoning in surgical videos through 1,123 doctor-grounded questions over self-built long neurosurgery recordings and a public lecture-video test split, with fixed evaluation dimensions for comparison.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Agents","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.14015","pdf":"https://arxiv.org/pdf/2608.14015","project":"https://fyycs.github.io/medclaw/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.14015"},"evidence":{"snippet":"To evaluate this agent, we introduce MedClawBench, a de-leaked, doctor-grounded benchmark of 1,123 questions over self-built long neurosurgery recordings and a held-out public lecture-video test split.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14015"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MedClawBench evaluates long-horizon temporal reasoning in surgical videos through 1,123 doctor-grounded questions over self-built long neurosurgery recordings and a public lecture-video test split, with fixed evaluation dimensions for comparison.","whyItMatters":"Existing VLM benchmarks fail to capture temporal dependencies in long surgical videos. MedClawBench provides a reproducible, doctor-grounded dataset to assess video reasoning capabilities beyond short clips, aiding progress in surgical AI systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7156f19c8562b3031ef6b9a38928e198f2688ff963fa2e2732adb09d9170c6a8"},"motivation":"Understanding tens-of-minutes surgical videos requires long-horizon temporal reasoning, answering what happens before, after, or across stages of a procedure by grounding the question in visual evidence spread across time.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14015","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MedClaw team","organizationType":"academic-lab","sourceUrl":"https://fyycs.github.io/medclaw/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_ec802ba96f54adeb","familyId":"catalog_family_ec802ba96f54adeb","name":"MedCode","oneLine":"Vals AI healthcare benchmark for whether models can support the medical billing process.","description":"Vals AI healthcare benchmark for whether models can support the medical billing process.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/medcode","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ec802ba96f54adeb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsmedcode"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsMedCode","url":"https://benchlm.ai/benchmarks/valsmedcode","paperUrl":"https://www.vals.ai/benchmarks/medcode","year":"2026","fullName":"Vals MedCode","format":"Accuracy score","tasks":"Medical billing support tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_medcta_84809bbe","familyId":"bmf_f4b829da8324","name":"MedCTA","oneLine":"MedCTA evaluates clinical tool agents on 107 clinician-verified, step-implicit tasks with multimodal inputs (radiology images, pathology slides, reports) and 5 deployed tools. Metrics cover tool selection, argument validity, execution stability, trajectory fidelity, and outcome quality.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11702","pdf":"https://arxiv.org/pdf/2606.11702","project":"https://ivul-kaust.github.io/MedCTA/","code":"https://github.com/IVUL-KAUST/MedCTA","data":"https://huggingface.co/datasets/IVUL-KAUST/MedCTA","hfPaper":"https://huggingface.co/papers/2606.11702"},"evidence":{"snippet":"We introduce MedCTA, a benchmark for evaluating medical tool agents on clinician-validated, step-implicit tasks grounded in realistic multimodal clinical inputs, including radiology images, pathology slides, and reports.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":465,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2606.11702"},"ranking":{"90d":{"score":37,"rank":168,"coverage":0.85,"confidence":"High","datasetDownloadRank":22,"datasetRankPopulation":66}},"description":"MedCTA evaluates clinical tool agents on 107 clinician-verified, step-implicit tasks with multimodal inputs (radiology images, pathology slides, reports) and 5 deployed tools. Metrics cover tool selection, argument validity, execution stability, trajectory fidelity, and outcome quality.","whyItMatters":"Fills a gap in medical AI evaluation by going beyond single-turn QA to test agentic tool use, planning, and reliability in real-world clinical workflows. Useful for auditing and advancing trustworthy medical AI agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1d3af7081fe9ba9810369fee5a9b4369e7e263d419ecf502a88211a080190291"},"motivation":"To make clinically grounded decisions, medical AI agents are expected to go beyond simple recognition and be capable of tool retrieval, evidence acquisition, and integration.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11702","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"IVUL-KAUST","organizationType":"academic-lab","sourceUrl":"https://ivul-kaust.github.io/MedCTA/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_medcua-bench_73e6da74","familyId":"bmf_928d146acf8d","name":"MedCUA-Bench","oneLine":"MedCUA-Bench is an interactive benchmark for clinical computer-use agents, covering 18 scenarios in 10 medical domains with deterministic safety evaluation.","area":"Agents & Tool Use","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03203","pdf":"https://arxiv.org/pdf/2606.03203","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03203"},"evidence":{"snippet":"We introduce MedCUA-Bench, an interactive benchmark for clinical computer-use agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03203"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MedCUA-Bench is an interactive benchmark for clinical computer-use agents, covering 18 scenarios in 10 medical domains with deterministic safety evaluation.","whyItMatters":"It highlights the gap in current agents' ability to operate clinical software, motivating safer and more reliable automation in healthcare.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b1fbb342fea81c8323b7b7b9ae9204b3ac2c105503b5eeaedc244458b0caa426"},"motivation":"Computer-use agents could automate repetitive screen-based clinical work, but their reliability in medical graphical user interfaces remains largely unvalidated.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03203","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_medddc-eval_84cc7de6","familyId":"bmf_7b4401a381b7","name":"MedDDC-Eval","oneLine":"MedDDC-Eval evaluates multi-turn medical consultation agents by decoupling diagnosis from the diagnostic reader, using a frozen shared reader to score policies. It reports diagnostic support, coverage, and efficiency across Record and Dialogue splits.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-07-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.18999","pdf":"https://arxiv.org/pdf/2607.18999","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.18999"},"evidence":{"snippet":"We introduce MedDDC-Eval, a diagnosis-decoupled evaluation testbed over held-out cases derived from medical records and online consultations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.18999"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MedDDC-Eval evaluates multi-turn medical consultation agents by decoupling diagnosis from the diagnostic reader, using a frozen shared reader to score policies. It reports diagnostic support, coverage, and efficiency across Record and Dialogue splits.","whyItMatters":"Coupled evaluation confounds history elicitation with terminal diagnosis. This testbed enables fair comparison and evaluation-informed policy optimization, providing a more reliable measure of diagnostic support.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"52783678b7ebb043c336d2412c93c1f7854b17231ee97d9c30a30cd8ffcdd4d1"},"motivation":"Evaluating multi-turn medical consultation agents requires judging the diagnostic support provided by the histories they elicit through interaction.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.18999","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_medfailbench_90f26faa","familyId":"bmf_539cc8579836","name":"MedFailBench","oneLine":"The evaluation object is unclear from the provided information.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Safety"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.15166","pdf":"https://arxiv.org/pdf/2607.15166","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.15166"},"evidence":{"snippet":"We present a synthetic benchmark and failure atlas built by a clinician.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.15166"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"The evaluation object is unclear from the provided information.","whyItMatters":"The evaluation gap and practical decision value are unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d16bb94e520321bc9eeeedb4468346730904d1df61f48a1b5bec2c0246ef335e"},"motivation":"Most medical AI benchmarks measure whether a model knows the correct answer.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.15166","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_medgym_d9be6475","familyId":"bmf_0a3477dc88fd","name":"MedGym","oneLine":"Provides a continuous-time reinforcement learning environment for dynamic medical treatment recommendation, built from clinical data with physics-informed neural networks. Supports offline and online RL with configurable parameters.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01028","pdf":"https://arxiv.org/pdf/2606.01028","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01028"},"evidence":{"snippet":"To address this gap, we introduce MedGym, a benchmark environment for dynamic treatment recommendation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01028"},"ranking":{},"description":"Provides a continuous-time reinforcement learning environment for dynamic medical treatment recommendation, built from clinical data with physics-informed neural networks. Supports offline and online RL with configurable parameters.","whyItMatters":"Offers a realistic testbed for RL methods in continuous-time medical settings, enabling evaluation of personalization, safety, and online deployment gaps. Standardizes comparisons between discrete and continuous time approaches.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"95a9aebde44e8244c4f90094756fbefb76d4dfde9aa581d721d7573e9a406ab4"},"motivation":"Medical treatment recommendation poses several challenges to reinforcement learning (RL): patient physiology evolves in continuous time, measurements and interventions are performed at irregular intervals, and treatment effects vary substantially across individuals.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01028","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MedGym Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.01028","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_medhal-loc_0d281208","familyId":"bmf_fec062af73af","name":"MedHal-Loc","oneLine":"MedHal-Loc is a benchmark and metric for localization faithfulness of medical hallucination detectors, comprising a controlled subset of 300 PubMedQA-derived statements with injected span-level errors and a natural subset. It evaluates whether top-ranked error units overlap erroneous spans across four error types.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-19","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.21517","pdf":"https://arxiv.org/pdf/2606.21517","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.21517"},"evidence":{"snippet":"We introduce MedHal-Loc, a benchmark and metric for localization faithfulness -- whether a detector's top-ranked error unit actually overlaps the erroneous span.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.21517"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MedHal-Loc is a benchmark and metric for localization faithfulness of medical hallucination detectors, comprising a controlled subset of 300 PubMedQA-derived statements with injected span-level errors and a natural subset. It evaluates whether top-ranked error units overlap erroneous spans across four error types.","whyItMatters":"The benchmark addresses the evaluation gap in measuring whether hallucination detectors that claim explainability actually localize errors faithfully, providing a way to assess detection and localization validity separately.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"05d36af4258e52c03953ae111f53149123a782f12f3952eb4b1fba5cc959ab6c"},"motivation":"Detecting hallucinations in clinical text is increasingly framed as an explainability problem: systems should not merely flag an unreliable response but point to the offending span.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.21517","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_medit-bench_8ae40735","familyId":"bmf_97c8c854879f","name":"MEDit-Bench","oneLine":"Evaluates message-driven narrative video editing with long-form videos paired with multiple editing messages and multiple professional edits per message, using temporal alignment metrics and additional annotations for message ambiguity and contextfulness.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25300","pdf":"https://arxiv.org/pdf/2607.25300","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.25300"},"evidence":{"snippet":"For evaluating message-driven video editing, we present \\textbf{MEDit-Bench}, a dataset and benchmark, which pairs long-form videos with multiple editing messages and multiple professionally produced edits per message, demonstrating that different messages yield substantially different edits from the same source.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25300"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates message-driven narrative video editing with long-form videos paired with multiple editing messages and multiple professional edits per message, using temporal alignment metrics and additional annotations for message ambiguity and contextfulness.","whyItMatters":"Addresses the gap in video editing evaluation by accounting for diverse editorial intents, providing a protocol to compare model and human performance on narrative-driven editing, and offering stratification by message difficulty.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6982c6768baa84a2a6ece02f27e26e54ce7fb6bc2895e5bb726ee658676c557a"},"motivation":"Video editing is fundamentally message-driven: even from the same source footage, the selected shots change depending on the narrative the editor wishes to convey.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25300","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_medlayxplain_c048bbf8","familyId":"bmf_befc27de9bc5","name":"MEDLAYXPLAIN","oneLine":"MedLayXPlain is a benchmark for medical lay language generation, pairing medical images with expert and lay captions across 122,789 samples from 8 imaging modalities. It introduces a 3B evaluator model scoring expert-lay alignment on five attributes.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-19","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.21194","pdf":"https://arxiv.org/pdf/2606.21194","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.21194"},"evidence":{"snippet":"To this end, we introduce MedLayXPlain, the first large-scale multimodal benchmark and evaluation framework for Medical Lay Language Generation (MLLG).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.21194"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MedLayXPlain is a benchmark for medical lay language generation, pairing medical images with expert and lay captions across 122,789 samples from 8 imaging modalities. It introduces a 3B evaluator model scoring expert-lay alignment on five attributes.","whyItMatters":"Addresses the gap between expert-level medical image descriptions and patient-accessible language, crucial for patient education and shared decision-making under recent regulations. Provides a standardized evaluation for medical VLMs in patient-facing communication.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"38408243427fa77658038347ccc33be48493991328a007a07c691952052e9496"},"motivation":"Medical Vision-Language Models (Med-VLMs) achieve strong expert-level performance, yet their ability to generate patient-accessible descriptions remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.21194","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_medpic-bench_f4184b8f","familyId":"bmf_f9677fb45e23","name":"MedPIC-Bench","oneLine":"MedPIC-Bench evaluates patient-specific medication-safety reasoning in LLMs using guideline-following and paired counterfactual questions, with 467 questions annotated across six dimensions.","area":"Safety & Trustworthiness","applicationDomains":["Cybersecurity","Health & Life Sciences"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity","Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Safety","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.03028","pdf":"https://arxiv.org/pdf/2608.03028","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03028"},"evidence":{"snippet":"To address this gap, we introduce MedPIC-Bench, a benchmark of source-verifiable recommendations and expert-validated questions for patient-specific medication-safety reasoning.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03028"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MedPIC-Bench evaluates patient-specific medication-safety reasoning in LLMs using guideline-following and paired counterfactual questions, with 467 questions annotated across six dimensions.","whyItMatters":"Static medication-safety accuracy may not reflect whether models use patient information to determine rule applicability. This benchmark aims to make conditional rule application measurable.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"20a8f713b6ed5f6b35987cf9f11377968ea22af6f25c06e64ea06d1ae59835cb"},"motivation":"Applying a valid medication-safety rule when its patient-specific conditions are not met can produce an incorrect decision.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03028","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"cross-domain"},{"id":"bm_medpress_3e59debc","familyId":"bmf_254263a6c1e2","name":"MedPRESS","oneLine":"MedPRESS evaluates LLM sycophancy in multi-turn medical dialogues, containing 600 five-turn scenarios across three families with structured judging and safety metrics.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02520","pdf":"https://arxiv.org/pdf/2608.02520","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.02520"},"evidence":{"snippet":"We introduce MedPRESS, a multi-turn benchmark for measuring patient-pressure-induced sycophancy in LLMs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02520"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MedPRESS evaluates LLM sycophancy in multi-turn medical dialogues, containing 600 five-turn scenarios across three families with structured judging and safety metrics.","whyItMatters":"Measures robustness under conversational pressure, a gap in static medical safety evaluations.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"27d4c7369310caf068cd25e943a2bf0b665f5ffa1f535771239004cb9edd6e24"},"motivation":"Large language models (LLMs) are increasingly used for health-related advice.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02520","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_c6a9517619666612","familyId":"catalog_family_c6a9517619666612","name":"MedQA","oneLine":"Evaluating language model bias in medical questions.","description":"Evaluating language model bias in medical questions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/medqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c6a9517619666612"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsmedqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsMedQa","url":"https://benchlm.ai/benchmarks/valsmedqa","paperUrl":"https://www.vals.ai/benchmarks/medqa","year":"2026","fullName":"Vals MedQA","format":"Accuracy score","tasks":"Medical question answering","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_medreamm_d93bafa9","familyId":"bmf_0483fab0d9cb","name":"MedReaMM","oneLine":"Evaluates LLMs on multimodal clinical diagnostic synthesis using 625 expert-validated cases with medical images and ICD-11 diagnoses.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.22323v1","pdf":"https://arxiv.org/pdf/2608.22323v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To bridge this gap, we introduce MedReaMM, a benchmark specifically designed to evaluate models' ability to synthesize heterogeneous clinical evidence consisting of detailed patient histories alongside multiple medical images into accurate differential diagnoses under a complete-information paradigm.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22323"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLMs on multimodal clinical diagnostic synthesis using 625 expert-validated cases with medical images and ICD-11 diagnoses.","whyItMatters":"Clinical diagnosis requires integrating heterogeneous evidence, and this benchmark highlights the gap between current models and expert-level synthesis.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"204237f110b7aa62d1ce2f18836a43e239fcce75952077adcda023cff7cf714d"},"motivation":"The application of Large Language Models (LLMs) to diagnostic decision-making has garnered growing interest.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The paper introduces a named benchmark with a curated case set and standardized diagnostic task, making it reusable for comparison.","canonicalNameSource":"paper_title","canonicalNameEvidence":"MedReaMM: Evaluating Large Multimodal Models on Expert-Level Clinical Diagnostic Synthesis"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22323v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":60,"confidence":"Medium","horizon":"7d","reason":"Medical multimodal reasoning is a high-impact area, and the expert-level focus may generate strong interest in the biomedical AI community."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_cbc25d0a19990a40","familyId":"catalog_family_cbc25d0a19990a40","name":"MedScribe","oneLine":"Vals AI healthcare benchmark for whether models can support doctors with administrative work.","description":"Vals AI healthcare benchmark for whether models can support doctors with administrative work.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/medscribe","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_cbc25d0a19990a40"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsmedscribe"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsMedScribe","url":"https://benchlm.ai/benchmarks/valsmedscribe","paperUrl":"https://www.vals.ai/benchmarks/medscribe","year":"2026","fullName":"Vals MedScribe","format":"Accuracy score","tasks":"Medical administrative support tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_medstreambench_73e2ea6c","familyId":"bmf_b4144a83ff7d","name":"MedStreamBench","oneLine":"MedStreamBench is a time-aware benchmark for medical video understanding, integrating 22 medical datasets and 5,419 QA instances across four temporal settings: retrospective, present, future, and proactive. Models are restricted to temporally bounded evidence windows and evaluated on answer correctness, responsiveness, and post-evidence stability.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.01751","pdf":"https://arxiv.org/pdf/2607.01751","project":null,"code":null,"data":"https://huggingface.co/datasets/Venn2024/MedStreamBench","hfPaper":"https://huggingface.co/papers/2607.01751"},"evidence":{"snippet":"We present MedStreamBench, a benchmark for time-aware medical video understanding.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":243,"hfDatasetLikes":4},"source":{"type":"arxiv","id":"2607.01751"},"ranking":{"90d":{"score":44,"rank":117,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":35,"datasetRankPopulation":66}},"description":"MedStreamBench is a time-aware benchmark for medical video understanding, integrating 22 medical datasets and 5,419 QA instances across four temporal settings: retrospective, present, future, and proactive. Models are restricted to temporally bounded evidence windows and evaluated on answer correctness, responsiveness, and post-evidence stability.","whyItMatters":"It addresses the gap between offline recognition and temporally grounded decision-making in clinical settings, where models must decide when to answer or alert. The benchmark provides a protocol for evaluating time-aware reasoning in medical video.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2d8a770354dccdbddfe653a361c7d0b86c6469f65b5308cb13be3d45b315e70d"},"motivation":"Existing medical video benchmarks primarily evaluate whether a model produces the correct answer, but rarely assess whether it answers at the right time.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01751","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_c72ce7241164b028","familyId":"catalog_family_c72ce7241164b028","name":"MedXpertQA","oneLine":"A comprehensive benchmark to evaluate expert-level medical knowledge and advanced reasoning, featuring 4,460 questions spanning 17 specialties and 11 body systems. Includes both text-only and multimodal subsets with expert-level exam questions incorporating diverse medical images and rich clinical information.","description":"A comprehensive benchmark to evaluate expert-level medical knowledge and advanced reasoning, featuring 4,460 questions spanning 17 specialties and 11 body systems. Includes both text-only and multimodal subsets with expert-level exam questions incorporating diverse medical images and rich clinical information.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Healthcare","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/medxpertqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c72ce7241164b028"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/medxpertqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"medxpertqa","url":"https://llm-stats.com/benchmarks/medxpertqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","healthcare","vision"],"catalogModelCount":12,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_33f4bd7a3190e5d9","familyId":"catalog_family_33f4bd7a3190e5d9","name":"MedXpertQA (MM)","oneLine":"MedXpertQA-MM is the multimodal subset of MedXpertQA, evaluating expert-level medical question answering grounded in medical images.","description":"MedXpertQA-MM is the multimodal subset of MedXpertQA, evaluating expert-level medical question answering grounded in medical images.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Knowledge","Medical","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_33f4bd7a3190e5d9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/medxpertqamm"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/medxpertqa-mm"}],"catalogSources":[{"catalog":"benchlm","sourceId":"medXpertQaMm","url":"https://benchlm.ai/benchmarks/medxpertqamm","paperUrl":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","year":"2026","fullName":"MedXpertQA Multimodal","format":"Medical visual MCQ","tasks":"2,000 multimodal medical questions","successorKey":null},{"catalog":"llm-stats","sourceId":"medxpertqa-mm","url":"https://llm-stats.com/benchmarks/medxpertqa-mm","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","knowledge","medical","multimodal","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_42409a88538592f7","familyId":"catalog_family_42409a88538592f7","name":"MedXpertQA (Text)","oneLine":"A medical multiple-choice benchmark spanning many specialties with 10 answer options per question.","description":"A medical multiple-choice benchmark spanning many specialties with 10 answer options per question.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_42409a88538592f7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/medxpertqatext"}],"catalogSources":[{"catalog":"benchlm","sourceId":"medXpertQaText","url":"https://benchlm.ai/benchmarks/medxpertqatext","paperUrl":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","year":"2026","fullName":"MedXpertQA Text","format":"Medical MCQ","tasks":"2,450 medical multiple-choice questions","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_meetingtom_9e241987","familyId":"bmf_69648c95f671","name":"MeetingToM","oneLine":"MeetingToM evaluates multimodal LLMs on theory-of-mind reasoning in multi-party meetings, including pseudo-consensus detection, across three levels: subject, dyad, and group. It provides a unified evaluation protocol.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19235","pdf":"https://arxiv.org/pdf/2607.19235","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19235"},"evidence":{"snippet":"We introduce MeetingToM, a benchmark for complex social behavior reasoning in naturalistic multi-party meetings.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19235"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MeetingToM evaluates multimodal LLMs on theory-of-mind reasoning in multi-party meetings, including pseudo-consensus detection, across three levels: subject, dyad, and group. It provides a unified evaluation protocol.","whyItMatters":"It covers latent social states and group dynamics often missing in existing ToM benchmarks, revealing limitations in integrating non-verbal cues and inferring hidden attitudes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"68b984db09633bd4ed37499e16247e584a24a1785d41080dc103e8dc04511726"},"motivation":"Theory of Mind (ToM), the ability to infer other's beliefs, intentions, and states of knowledge, is central to social interaction, yet remains challenging for current Multimodal Large Language Models (MLLMs), especially in multi-party meetings where cues are distributed across speech and behavior.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19235","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_8b815b926ea522f8","familyId":"catalog_family_8b815b926ea522f8","name":"MEGA MLQA","oneLine":"MLQA as part of the MEGA (Multilingual Evaluation of Generative AI) benchmark suite. A multi-way aligned extractive QA evaluation benchmark for cross-lingual question answering across 7 languages (English, Arabic, German, Spanish, Hindi, Vietnamese, and Simplified Chinese) with over 12K QA instances in English and 5K in each other language.","description":"MLQA as part of the MEGA (Multilingual Evaluation of Generative AI) benchmark suite. A multi-way aligned extractive QA evaluation benchmark for cross-lingual question answering across 7 languages (English, Arabic, German, Spanish, Hindi, Vietnamese, and Simplified Chinese) with over 12K QA instances in English and 5K in each other language.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mega-mlqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8b815b926ea522f8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mega-mlqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mega-mlqa","url":"https://llm-stats.com/benchmarks/mega-mlqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_c079a6f3139250aa","familyId":"catalog_family_c079a6f3139250aa","name":"MEGA TyDi QA","oneLine":"TyDi QA as part of the MEGA benchmark suite. A question answering dataset covering 11 typologically diverse languages (Arabic, Bengali, English, Finnish, Indonesian, Japanese, Korean, Russian, Swahili, Telugu, and Thai) with 204K question-answer pairs. Features realistic information-seeking questions written by people who want to know the answer but don't know it yet.","description":"TyDi QA as part of the MEGA benchmark suite. A question answering dataset covering 11 typologically diverse languages (Arabic, Bengali, English, Finnish, Indonesian, Japanese, Korean, Russian, Swahili, Telugu, and Thai) with 204K question-answer pairs. Features realistic information-seeking questions written by people who want to know the answer but don't know it yet.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mega-tydi-qa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c079a6f3139250aa"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mega-tydi-qa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mega-tydi-qa","url":"https://llm-stats.com/benchmarks/mega-tydi-qa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d9dab411d32bb30d","familyId":"catalog_family_d9dab411d32bb30d","name":"MEGA UDPOS","oneLine":"Universal Dependencies POS tagging as part of the MEGA benchmark suite. A multilingual part-of-speech tagging dataset based on Universal Dependencies treebanks, utilizing the universal POS tag set of 17 tags across 38 diverse languages from different language families. Used for evaluating multilingual POS tagging systems.","description":"Universal Dependencies POS tagging as part of the MEGA benchmark suite. A multilingual part-of-speech tagging dataset based on Universal Dependencies treebanks, utilizing the universal POS tag set of 17 tags across 38 diverse languages from different language families. Used for evaluating multilingual POS tagging systems.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mega-udpos","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d9dab411d32bb30d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mega-udpos"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mega-udpos","url":"https://llm-stats.com/benchmarks/mega-udpos","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_bda51ce98572a0ed","familyId":"catalog_family_bda51ce98572a0ed","name":"MEGA XCOPA","oneLine":"XCOPA (Cross-lingual Choice of Plausible Alternatives) as part of the MEGA benchmark suite. A typologically diverse multilingual dataset for causal commonsense reasoning in 11 languages, including resource-poor languages like Eastern Apurímac Quechua and Haitian Creole. Requires models to select which choice is the effect or cause of a given premise.","description":"XCOPA (Cross-lingual Choice of Plausible Alternatives) as part of the MEGA benchmark suite. A typologically diverse multilingual dataset for causal commonsense reasoning in 11 languages, including resource-poor languages like Eastern Apurímac Quechua and Haitian Creole. Requires models to select which choice is the effect or cause of a given premise.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mega-xcopa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bda51ce98572a0ed"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mega-xcopa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mega-xcopa","url":"https://llm-stats.com/benchmarks/mega-xcopa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_93ac083d8fb4e9cd","familyId":"catalog_family_93ac083d8fb4e9cd","name":"MEGA XStoryCloze","oneLine":"XStoryCloze as part of the MEGA benchmark suite. A cross-lingual story completion task that consists of professionally translated versions of the English StoryCloze dataset to 10 non-English languages. Requires models to predict the correct ending for a given four-sentence story, evaluating commonsense reasoning and narrative understanding.","description":"XStoryCloze as part of the MEGA benchmark suite. A cross-lingual story completion task that consists of professionally translated versions of the English StoryCloze dataset to 10 non-English languages. Requires models to predict the correct ending for a given four-sentence story, evaluating commonsense reasoning and narrative understanding.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mega-xstorycloze","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_93ac083d8fb4e9cd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mega-xstorycloze"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mega-xstorycloze","url":"https://llm-stats.com/benchmarks/mega-xstorycloze","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_cfe6a6676d638dfd","familyId":"catalog_family_cfe6a6676d638dfd","name":"Meld","oneLine":"MELD (Multimodal EmotionLines Dataset) is a multimodal multi-party dataset for emotion recognition in conversations. Contains approximately 13,000 utterances from 1,433 dialogues extracted from the TV series Friends. Each utterance is annotated with emotion (Anger, Disgust, Sadness, Joy, Neutral, Surprise, Fear) and sentiment labels across audio, visual, and textual modalities.","description":"MELD (Multimodal EmotionLines Dataset) is a multimodal multi-party dataset for emotion recognition in conversations. Contains approximately 13,000 utterances from 1,433 dialogues extracted from the TV series Friends. Each utterance is annotated with emotion (Anger, Disgust, Sadness, Joy, Neutral, Surprise, Fear) and sentiment labels across audio, visual, and textual modalities.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Psychology","Creativity"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/meld","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_cfe6a6676d638dfd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/meld"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"meld","url":"https://llm-stats.com/benchmarks/meld","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","psychology","creativity"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_memebench_baa40d97","familyId":"bmf_6bf115ae11a0","name":"MemeBench","oneLine":"MemeBench is a diagnostic benchmark of 1,253 Chinese and English memes with human-written references and VIKR annotations—Visual clues, Identity links, Knowledge units, and Reasoning mechanisms—for evaluating interpretation in LVLMs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27798","pdf":"https://arxiv.org/pdf/2607.27798","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.27798"},"evidence":{"snippet":"We introduce MemeBench, a diagnostic benchmark of 1,253 Chinese and English memes with human-written references and quality-controlled VIKR annotations, centered on anime, comics, games, and adjacent online subcultures.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27798"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MemeBench is a diagnostic benchmark of 1,253 Chinese and English memes with human-written references and VIKR annotations—Visual clues, Identity links, Knowledge units, and Reasoning mechanisms—for evaluating interpretation in LVLMs.","whyItMatters":"Memes rely on cultural knowledge beyond visual content, and this benchmark attempts to decompose interpretation into components. It could help identify gaps in LVLM understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a249a484afb479f5ad284f292969923fcbab754df104ef3581f8bdb734f0fce0"},"motivation":"Large vision-language models have improved at describing visual content, but accurate descriptions do not ensure interpretation when meaning depends on knowledge beyond the pixels.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27798","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_memfail_db6b29b0","familyId":"bmf_c88fa7965f13","name":"MemFail","oneLine":"Diagnostic benchmark isolating failure modes of LLM memory systems by evaluating summarization, storage, and retrieval operations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26667","pdf":"https://arxiv.org/pdf/2605.26667","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.26667"},"evidence":{"snippet":"We introduce MemFail, a diagnostic benchmark that isolates the failure modes of modern LLM memory systems.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26667"},"ranking":{},"description":"Diagnostic benchmark isolating failure modes of LLM memory systems by evaluating summarization, storage, and retrieval operations.","whyItMatters":"Existing memory benchmarks report aggregate QA accuracy, failing to attribute errors to specific system components. MemFail provides fine-grained diagnostics for memory system design.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"69c64a9638dd7f535ba6c023ef9e4962fb1eed32907a3c419885b0517366e7e1"},"motivation":"Large language model (LLM) agents increasingly rely on external memory systems to remain consistent across long-horizon interactions, but little empirical work has been done to understand the specific failure modes and design choices that these systems present.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26667","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_memobench_def5688a","familyId":"bmf_e8d56608be68","name":"MemoBench","oneLine":"MemoBench evaluates world modeling in video generation via a disappear-and-reappear paradigm. It includes 360 ground-truth clips and an evaluation suite with automated metrics and VQA across four diagnostic pillars.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.27537","pdf":"https://arxiv.org/pdf/2606.27537","project":null,"code":"https://github.com/MemoBench-Team/MemoBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.27537"},"evidence":{"snippet":"To bridge this gap, we introduce MemoBench, a diagnostic benchmark built around the disappear-and-reappear paradigm in dynamically changing environments: a target object undergoes a physical process, disappears from view, and must be correctly recovered in its updated state upon reappearance.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":6,"hfDailySubmittedAt":"2026-06-29T00:00:00.000Z","githubStars":40,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.27537"},"ranking":{"90d":{"score":48,"rank":85,"coverage":0.7,"confidence":"Medium"}},"description":"MemoBench evaluates world modeling in video generation via a disappear-and-reappear paradigm. It includes 360 ground-truth clips and an evaluation suite with automated metrics and VQA across four diagnostic pillars.","whyItMatters":"Current benchmarks assess memory consistency only when objects remain visible; MemoBench tests object recovery after occlusion in dynamic scenes, providing a reusable diagnostic for world models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bcf200e2fffb5ff18b81ee5d86bcf58665a22562b7be9ebf498b18c326981ff2"},"motivation":"Video generation models aspire to simulate dynamic environments, and several benchmarks now evaluate memory consistency across frames.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.27537","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MemoBench Team","organizationType":"community","sourceUrl":"https://github.com/MemoBench-Team/MemoBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_memops_f8d37e23","familyId":"bmf_1d0b276100ed","name":"MemOps","oneLine":"MemOps evaluates conversational memory as a sequence of lifecycle operations (remembering, forgetting, updating, reflecting) with structured traces and six categories of operation-level probes, under adjacent-evidence and long-context settings.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.12893","pdf":"https://arxiv.org/pdf/2607.12893","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.12893"},"evidence":{"snippet":"We introduce MemOps, a benchmark that reformulates conversational memory as a sequence of lifecycle operations and represents each memory event with a structured trace specifying its trigger, target, scope, state transition, and supporting evidence.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.12893"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MemOps evaluates conversational memory as a sequence of lifecycle operations (remembering, forgetting, updating, reflecting) with structured traces and six categories of operation-level probes, under adjacent-evidence and long-context settings.","whyItMatters":"This benchmark addresses the gap in memory evaluation by providing operation-level diagnosis rather than final-answer accuracy, revealing specific failure modes in long-context, retrieval-based, parametric, and managed-memory systems, which is valuable for improving memory reliability in LLM agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"000d346270f70feb830ce9c035e05914550c4063c14ea44731362f56401b2ce4"},"motivation":"Long-term memory has become a foundational capability for LLM-based agents that accompany users across extended, multi-session interactions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.12893","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_memorydocdataset_b8a1ebc4","familyId":"bmf_19a123656f60","name":"MemoryDocDataSet","oneLine":"MemoryDocDataSet evaluates joint conversational memory and long-document reasoning through 50 synthetic micro-worlds, each with personas, temporal event graphs, real legal documents, multi-session conversations, and QA pairs. Questions are categorized by reasoning type, with Hybrid questions requiring navigation of conversation history to locate the relevant document and extract answers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04442","pdf":"https://arxiv.org/pdf/2606.04442","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04442"},"evidence":{"snippet":"We introduce MemoryDocDataSet, a synthetic benchmark of 50 micro-worlds and 1,000 QA pairs in which each instance comprises 3-5 personas, a temporal event graph spanning months of activity, 3-5 real long documents (20,000-50,000 tokens each sourced from the Caselaw Access Project), multi-session conversations grounded on those documents, and 20 question-answer pairs across five reasoning categories.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04442"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MemoryDocDataSet evaluates joint conversational memory and long-document reasoning through 50 synthetic micro-worlds, each with personas, temporal event graphs, real legal documents, multi-session conversations, and QA pairs. Questions are categorized by reasoning type, with Hybrid questions requiring navigation of conversation history to locate the relevant document and extract answers.","whyItMatters":"Existing benchmarks evaluate conversational memory or document reasoning separately, leaving a gap in measuring integrated performance. MemoryDocDataSet provides a controlled, synthetic environment to assess systems on tasks that require both capabilities, useful for developing and comparing architectures that unify conversation memory with long-document retrieval.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7582237fc995bda61f2d12f50b43f09493628f139ae5c218d4e2abd742f21d33"},"motivation":"AI systems increasingly need to combine two demanding capabilities: navigating multi-session conversation history and performing deep reading comprehension within long documents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04442","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_memprobe_9e043327","familyId":"bmf_8b0112f7dc9d","name":"MEMPROBE","oneLine":"MEMPROBE evaluates long-term memory in LLM agents by reconstructing hidden user-state from agent memory across 50 simulated users with 31 hidden dimensions each.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.24595","pdf":"https://arxiv.org/pdf/2606.24595","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.24595"},"evidence":{"snippet":"We instantiate this view in MEMPROBE, a benchmark in which a memory-equipped agent assists simulated users, each carrying a hidden, taxonomy-anchored user-state bank, across a trajectory of leak-controlled tasks, after which that bank is reconstructed from the agent's resulting memory under both full-store and top-k access.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":"2026-06-24T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24595"},"ranking":{"90d":{"score":45,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MEMPROBE evaluates long-term memory in LLM agents by reconstructing hidden user-state from agent memory across 50 simulated users with 31 hidden dimensions each.","whyItMatters":"Offers a direct measure of memory fidelity as an auditable artifact, distinct from downstream task success, potentially improving agent memory evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e3b0d2b5d0a4c211b19e81ca11157d54d040ed43f738876f5848b936945f3209"},"motivation":"Long-term memory promises LLM agents that grow more capable across sessions, maintaining an accurate, evolving understanding of the user that interaction forms.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24595","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_memsecbench_ec92c168","familyId":"bmf_d73b74353527","name":"MemSecBench","oneLine":"Evaluates lifecycle security of agent memory systems with 310 cases across 48 contexts, using a Write-Execute-Forget protocol and evidence-based adjudication across seven checkpoints.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27080","pdf":"https://arxiv.org/pdf/2607.27080","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.27080"},"evidence":{"snippet":"To address this gap, we introduce MemSecBench, a task-grounded benchmark for the lifecycle security of agent memory systems.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27080"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates lifecycle security of agent memory systems with 310 cases across 48 contexts, using a Write-Execute-Forget protocol and evidence-based adjudication across seven checkpoints.","whyItMatters":"Provides insight into how malicious instructions can persist in memory systems and affect later actions, highlighting security differences across configurations.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"019db75a2085515681bf1d13548a7b136775d006cf368ae4f9dd9963d73e8796"},"motivation":"Memory systems allow agents to retain and reuse information from past interactions, but they can also let malicious content persist.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27080","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_memsyco-bench_f1a6a168","familyId":"bmf_3d36e1cca960","name":"MemSyco-Bench","oneLine":"MemSyco-Bench evaluates memory-induced sycophancy in agent systems across five tasks, measuring how memory influences decision-making and personalization.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.01071","pdf":"https://arxiv.org/pdf/2607.01071","project":null,"code":"https://github.com/XMUDeepLIT/MemSyco-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.01071"},"evidence":{"snippet":"To bridge this gap, we propose MemSyco-Bench, a comprehensive benchmark for evaluating memory-induced sycophancy in agent systems.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":31,"hfDailySubmittedAt":"2026-07-02T00:00:00.000Z","githubStars":17,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01071"},"ranking":{"90d":{"score":46,"rank":99,"coverage":0.7,"confidence":"Medium"}},"description":"MemSyco-Bench evaluates memory-induced sycophancy in agent systems across five tasks, measuring how memory influences decision-making and personalization.","whyItMatters":"Addresses a gap in memory benchmarks by focusing on downstream reasoning effects, offering a leaderboard and standardized evaluation code.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"73a3d5d893eb00c0bf467fe671cc3ad6247bd0f6ac298ddfc89ed82a5f1bd692"},"motivation":"Memory has emerged as a cornerstone of modern LLM-based agents, supporting their evolution from single-turn assistants to long-term collaborators.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01071","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"XMUDeepLIT","organizationType":"academic-lab","sourceUrl":"https://github.com/XMUDeepLIT/MemSyco-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_memtrace_56029435","familyId":"bmf_cc4e1edaadda","name":"MemTrace","oneLine":"MemTrace evaluates long-term memory in LLM agents at the knowledge point level, probing memory age, question type, and evidence condition. Scoring is based on accuracy across these dimensions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.17328","pdf":"https://arxiv.org/pdf/2606.17328","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.17328"},"evidence":{"snippet":"We introduce MemTrace, a benchmark whose unit of measurement is the knowledge point: a single typed fact about the user, rather than an individual question.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.17328"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MemTrace evaluates long-term memory in LLM agents at the knowledge point level, probing memory age, question type, and evidence condition. Scoring is based on accuracy across these dimensions.","whyItMatters":"Aggregated accuracy misses important memory behaviors. MemTrace provides finer-grained metrics to reveal bottlenecks in evidence use, guiding improvements in memory systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"337defe4b68c90141b6c7e584e0aeee3beabd9ddf9ed8399eaa29a648260620d"},"motivation":"LLM agents increasingly maintain long-term memory of user facts across sessions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.17328","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_meniomni_7f0a7094","familyId":"bmf_7f40c7d5cfda","name":"MeniOmni","oneLine":"MeniOmni is a multimodal benchmark for meniscus injury assessment with 746 MRI studies, supporting Stoller severity grading and diagnostic report generation, with risk-aware ordinal evaluation and semantic consistency metric.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.28161","pdf":"https://arxiv.org/pdf/2605.28161","project":null,"code":"https://github.com/ShuruiXu/MeniOmni","data":null,"hfPaper":"https://huggingface.co/papers/2605.28161"},"evidence":{"snippet":"We introduce MeniOmni, a structured multimodal benchmark for meniscus injury assessment, consisting of 746 multi-center MRI studies with tri-planar volumetric inputs, Clinical Priors, and expert-annotated clinical text.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28161"},"ranking":{},"description":"MeniOmni is a multimodal benchmark for meniscus injury assessment with 746 MRI studies, supporting Stoller severity grading and diagnostic report generation, with risk-aware ordinal evaluation and semantic consistency metric.","whyItMatters":"Addresses a gap in knee MRI benchmarks by integrating volumetric MRI data with clinical context for holistic clinical reasoning, aiming to improve safety in meniscus injury diagnosis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"86f783e39c54113133330e0e018d95c1ac29899bc47185a0cd2a433602f79f19"},"motivation":"Clinical diagnosis of meniscus injuries requires radiologists to integrate volumetric MRI evidence with patient context (e.g., sex, age, BMI) and to produce structured diagnostic reports.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"IEEE International Conference on Multimedia and Expo (ICME) 2026 (Oral Presentation)","evidence":"Accepted by IEEE International Conference on Multimedia and Expo (ICME) 2026 (Oral Presentation)","evidenceUrl":"https://arxiv.org/abs/2605.28161","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"IEEE International Conference on Multimedia and Expo (ICME) 2026 (Oral Presentation)","reviewStatus":"accepted","decisionRaw":"Accepted by IEEE International Conference on Multimedia and Expo (ICME) 2026 (Oral Presentation)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2605.28161","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by IEEE International Conference on Multimedia and Expo (ICME) 2026 (Oral Presentation)","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_mergemedbench_c8868c16","familyId":"bmf_2eaf0e9f9c36","name":"MergeMedBench","oneLine":"MergeMedBench evaluates model merging methods for medical LVLMs across eight imaging modalities, comprising 16 LoRA fine-tuned models. It provides evaluation datasets and released model checkpoints for benchmarking merging approaches.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.15661","pdf":"https://arxiv.org/pdf/2607.15661","project":null,"code":"https://github.com/MedAI-T/MergeMedBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.15661"},"evidence":{"snippet":"We introduce MergeMedBench, a comprehensive benchmark spanning eight imaging modalities and diverse clinical task types, comprising 16 LoRA fine-tuned models built upon two mainstream architectures.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.15661"},"ranking":{"90d":{"score":23,"rank":371,"coverage":0.55,"confidence":"Low"}},"description":"MergeMedBench evaluates model merging methods for medical LVLMs across eight imaging modalities, comprising 16 LoRA fine-tuned models. It provides evaluation datasets and released model checkpoints for benchmarking merging approaches.","whyItMatters":"Addresses the lack of systematic evaluation for model merging in medical imaging, enabling comparison of merging methods and serving as a practical baseline.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"666a80260bf52cd62c7378756cf128bd43fbe792867a4b80ab62254e3842d030"},"motivation":"Large vision-language models (LVLMs) can be adapted to specialized medical imaging tasks via parameter-efficient fine-tuning approaches such as low-rank adaptation (LoRA), leading to a growing ecosystem of expert models tailored to specific imaging modalities and clinical scenarios.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.15661","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_4df8055b7a28417d","familyId":"catalog_family_4df8055b7a28417d","name":"Meta Internal Coding Bench","oneLine":"Meta's internal evaluation of coding-agent performance.","description":"Meta's internal evaluation of coding-agent performance.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/meta-internal-coding-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4df8055b7a28417d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/meta-internal-coding-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"meta-internal-coding-bench","url":"https://llm-stats.com/benchmarks/meta-internal-coding-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_metabonet-bench_963d2082","familyId":"bmf_7692d167ab51","name":"MetaboNet-Bench","oneLine":"MetaboNet-Bench evaluates multimodal glucose forecasting in type 1 diabetes, using glucose, insulin, and carbohydrate data. It provides an extensible open-source evaluation framework for comparing forecasting algorithms.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.18640","pdf":"https://arxiv.org/pdf/2606.18640","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18640"},"evidence":{"snippet":"Here, we introduce MetaboNet-Bench, a benchmark for multimodal glucose forecasting for patients with type 1 diabetes that provides an extensible open-source evaluation framework for comparison of glucose forecasting algorithms that leverage glucose, insulin, and carbohydrate data.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18640"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MetaboNet-Bench evaluates multimodal glucose forecasting in type 1 diabetes, using glucose, insulin, and carbohydrate data. It provides an extensible open-source evaluation framework for comparing forecasting algorithms.","whyItMatters":"Standardized benchmarks are lacking in glucose forecasting, hindering fair comparison. MetaboNet-Bench addresses this by offering a multimodal framework, enabling assessment of algorithms that leverage multiple data signals.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3c940c0064d827a6e8a5f1287ef946a80191ee189216ca3bc3ed4fcc3acb55c9"},"motivation":"Glucose forecasting algorithms are an important aspect of glycemic control management in type 1 diabetes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18640","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_metaphorvu-bench_ab9e08cf","familyId":"bmf_2f1827922297","name":"MetaphorVU-Bench","oneLine":"MetaphorVU-Bench evaluates metaphorical video understanding in multimodal LLMs through tasks requiring cross-domain mapping, with a benchmark dataset and evaluation code publicly available.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.25461","pdf":"https://arxiv.org/pdf/2605.25461","project":null,"code":"https://github.com/icip-cas/MetaphorVU","data":null,"hfPaper":"https://huggingface.co/papers/2605.25461"},"evidence":{"snippet":"To bridge this gap, we propose MetaphorVU-Bench, the first systematic and comprehensive benchmark dedicated to metaphorical video understanding.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":null,"githubStars":10,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25461"},"ranking":{},"description":"MetaphorVU-Bench evaluates metaphorical video understanding in multimodal LLMs through tasks requiring cross-domain mapping, with a benchmark dataset and evaluation code publicly available.","whyItMatters":"It fills the gap of systematic evaluation of high-order cognitive capabilities in video understanding, revealing defects in cross-domain mapping and enabling future improvements.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d8c0f822677392ee51b5ed37bebe65731cf5117faab044fd4dd0e314e22ce090"},"motivation":"Metaphorical videos are prevalent across various real-world scenarios to convey complex ideas, and understanding them typically requires high-order cognitive capabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25461","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MetaphorVU team","organizationType":"academic-lab","sourceUrl":"https://github.com/icip-cas/MetaphorVU","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_metaroute-bench_a64b1073","familyId":"bmf_fd31642f325d","name":"MetaRoute-Bench","oneLine":"MetaRoute-Bench is an open framework for comparing meta-decision policies in agentic workflow routing. Contains 180 synthetic task profiles, eight routing policies, and 30 seeds, evaluating success, cost, and latency through 43,200 traces. Metrics include success rate and paired confidence intervals.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00107","pdf":"https://arxiv.org/pdf/2608.00107","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00107"},"evidence":{"snippet":"We present MetaRoute-Bench, an open, inspectable framework for comparing meta-decision policies under a shared execution model.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00107"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MetaRoute-Bench is an open framework for comparing meta-decision policies in agentic workflow routing. Contains 180 synthetic task profiles, eight routing policies, and 30 seeds, evaluating success, cost, and latency through 43,200 traces. Metrics include success rate and paired confidence intervals.","whyItMatters":"Meta-decisions in agentic systems are often embedded and evaluated only via aggregate task accuracy. MetaRoute-Bench provides a shared execution model to compare routing policies, enabling analysis of tradeoffs between success, cost, and latency.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b26f1040e6a5316c1e7822e89434fcac4045ac80b2665ade155a93efcd9bedd2"},"motivation":"Agentic systems must repeatedly decide whether to answer directly, decompose a task, invoke a tool, execute code, delegate to a specialist, verify an intermediate result, or recover from failure.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00107","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_meventbench_dad2f997","familyId":"bmf_bdb6464af549","name":"MEventBench","oneLine":"MEventBench is a multi-event long video understanding benchmark introduced alongside the MoD-VLLM framework, evaluating temporal grounding and semantic understanding.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.15778","pdf":"https://arxiv.org/pdf/2607.15778","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.15778"},"evidence":{"snippet":"Moreover, we propose MEventBench, a challenging Multi-Event Long Video Benchmark for complex long video reasoning.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.15778"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MEventBench is a multi-event long video understanding benchmark introduced alongside the MoD-VLLM framework, evaluating temporal grounding and semantic understanding.","whyItMatters":"Could support evaluation of long-video understanding, but primarily serves as a testbed for the proposed method.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4446152766221e78ac06b96b8455613f180321d81d48c3bbfd04420ee9e51804"},"motivation":"Video Large Language Models (Video LLMs) have made significant advancements in various video understanding tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"2026 IEEE International Conference on Multimedia and Expo (ICME 2026)","evidence":"Accepted by 2026 IEEE International Conference on Multimedia and Expo (ICME 2026)","evidenceUrl":"https://arxiv.org/abs/2607.15778","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"2026 IEEE International Conference on Multimedia and Expo (ICME 2026)","reviewStatus":"accepted","decisionRaw":"Accepted by 2026 IEEE International Conference on Multimedia and Expo (ICME 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.15778","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by 2026 IEEE International Conference on Multimedia and Expo (ICME 2026)","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_19459b4c43b239f7","familyId":"catalog_family_19459b4c43b239f7","name":"MEWC","oneLine":"MEWC is a benchmark that evaluates AI model performance on multi-environment web challenges, testing agents' ability to navigate and complete complex tasks across diverse web environments.","description":"MEWC is a benchmark that evaluates AI model performance on multi-environment web challenges, testing agents' ability to navigate and complete complex tasks across diverse web environments.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.minimax.io/news/minimax-m25","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_19459b4c43b239f7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mewc"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mewc"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mewc","url":"https://benchlm.ai/benchmarks/mewc","paperUrl":"https://www.minimax.io/news/minimax-m25","year":"2026","fullName":"Multi-Environment Web Challenge","format":"Browser task completion","tasks":"Web-agent tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"mewc","url":"https://llm-stats.com/benchmarks/mewc","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_92041218cde9f5c1","familyId":"catalog_family_92041218cde9f5c1","name":"MGSM","oneLine":"MGSM (Multilingual Grade School Math) is a benchmark of grade-school math problems. Contains 250 grade-school math problems manually translated from the GSM8K dataset into ten typologically diverse languages: Spanish, French, German, Russian, Chinese, Japanese, Thai, Swahili, Bengali, and Telugu. Evaluates multilingual mathematical reasoning capabilities.","description":"MGSM (Multilingual Grade School Math) is a benchmark of grade-school math problems. Contains 250 grade-school math problems manually translated from the GSM8K dataset into ten typologically diverse languages: Spanish, French, German, Russian, Chinese, Japanese, Thai, Swahili, Bengali, and Telugu. Evaluates multilingual mathematical reasoning capabilities.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multilingual","External","Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2210.03057","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_92041218cde9f5c1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mgsm"},{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsmgsm"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mgsm"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mgsm","url":"https://benchlm.ai/benchmarks/mgsm","paperUrl":"https://arxiv.org/abs/2210.03057","year":"2022","fullName":"Multilingual Grade School Math","format":"Math word problems","tasks":"250 problems × 11 languages","successorKey":null},{"catalog":"benchlm","sourceId":"valsMgsm","url":"https://benchlm.ai/benchmarks/valsmgsm","paperUrl":"https://www.vals.ai/benchmarks/mgsm","year":"2026","fullName":"Vals MGSM","format":"Accuracy score","tasks":"Multilingual grade-school math questions","successorKey":null},{"catalog":"llm-stats","sourceId":"mgsm","url":"https://llm-stats.com/benchmarks/mgsm","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multilingual","external","math","reasoning"],"catalogModelCount":31,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_when-names-cross-scripts-a-source-grounded_350cc333","familyId":"bmf_1a9235cff5b5","name":"MHER","oneLine":"A provenance-controlled benchmark for historical entity reconciliation, comprising a Name-only core of 396 pairs and a Source-grounded subset of 160 pairs with entity-disjoint splits.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":0.45,"links":{"report":"http://arxiv.org/abs/2608.23507v1","pdf":"https://arxiv.org/pdf/2608.23507v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce MHER, a provenance-controlled benchmark for pairwise reconciliation of person-name attestations from the Mongol world.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23507"},"ranking":{"today":{"score":42,"rank":17,"coverage":0.4,"confidence":"Low"},"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A provenance-controlled benchmark for historical entity reconciliation, comprising a Name-only core of 396 pairs and a Source-grounded subset of 160 pairs with entity-disjoint splits.","whyItMatters":"Provides controlled framework to study evidence use and failure modes in historical NLP, showing that source-grounded evidence substantially improves identity resolution over names alone.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"924efe094d5eb82a5cc1b26fa42cefcd1d75cc013ff79fe7d11cc4e87e34bfcc"},"motivation":"Historical people may appear under different languages, scripts, and transcription traditions, while distinct individuals may share highly similar or even identical names.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark defines a stable pairwise evaluation protocol with entity-disjoint splits, but no explicit public artifact link or release statement is provided.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce MHER, a provenance-controlled benchmark for pairwise reconciliation"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.23507v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":30,"confidence":"Low","horizon":"7d","reason":"Niche historical NLP topic with limited breadth; without public artifacts, early visibility is restrained."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0e92240721888416","familyId":"catalog_family_0e92240721888416","name":"MIABench","oneLine":"MIABench evaluates multimodal instruction alignment and following capabilities.","description":"MIABench evaluates multimodal instruction alignment and following capabilities.","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Instruction Following","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/miabench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0e92240721888416"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/miabench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"miabench","url":"https://llm-stats.com/benchmarks/miabench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["instruction following","multimodal","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general"},{"id":"bm_migue-bench_dbd75fa4","familyId":"bmf_9437ac27c8a9","name":"MiGUE-Bench","oneLine":"MiGUE-Bench is a benchmark for multi-granularity event analysis, covering event detection, relation reasoning, structure induction, and future prediction across single to cross-document settings. It uses an LLM-driven annotation pipeline.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27654","pdf":"https://arxiv.org/pdf/2607.27654","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.27654"},"evidence":{"snippet":"To address these limitations, we introduce MiGUE-Bench, a systematic benchmark for assessing the performance of LLMs in multi-granularity event analysis.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27654"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MiGUE-Bench is a benchmark for multi-granularity event analysis, covering event detection, relation reasoning, structure induction, and future prediction across single to cross-document settings. It uses an LLM-driven annotation pipeline.","whyItMatters":"Event analysis spans diverse tasks at different document granularity, and this benchmark aims to systematically assess LLM capabilities across them. Its breadth could inform progress in information extraction.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9ac34bd126caeaece8c760ab62c2a66247d8eb556cb6efaf9322ef7e2da1c7aa"},"motivation":"Event analysis is an essential and fundamental direction of information extraction, involving various event-centric tasks at different granularity of documents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"publication_reported","venue":"Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR '26), pp. 3464-3472, 2026","evidence":"Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR '26), pp. 3464-3472, 2026","evidenceUrl":"https://arxiv.org/abs/2607.27654","source":"arxiv-journal-reference","evidenceLevel":"strong-author-metadata","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publications":[{"venueName":"Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR '26), pp. 3464-3472, 2026","publicationStatus":"published","evidence":[{"sourceType":"arxiv-journal-reference","sourceUrl":"https://arxiv.org/abs/2607.27654","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR '26), pp. 3464-3472, 2026","level":"strong-author-metadata"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_073f18f74425caec","familyId":"catalog_family_073f18f74425caec","name":"MILU","oneLine":"Culturally grounded knowledge comprehension across ten Indic languages and English.","description":"Culturally grounded knowledge comprehension across ten Indic languages and English.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multilingual"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2411.02538","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_073f18f74425caec"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/milu"}],"catalogSources":[{"catalog":"benchlm","sourceId":"milu","url":"https://benchlm.ai/benchmarks/milu","paperUrl":"https://arxiv.org/abs/2411.02538","year":"2024","fullName":"Multi-task Indic Language Understanding Benchmark","format":"Average accuracy","tasks":"Knowledge tasks across 11 languages","successorKey":null}],"catalogCategories":["multilingual"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_8fad0036a02a982d","familyId":"catalog_family_8fad0036a02a982d","name":"MIMIC CXR","oneLine":"MIMIC-CXR is a large publicly available dataset of chest radiographs with free-text radiology reports. Contains 377,110 images corresponding to 227,835 radiographic studies from 65,379 patients at Beth Israel Deaconess Medical Center. The dataset is de-identified and widely used for medical imaging research, automated report generation, and medical AI development.","description":"MIMIC-CXR is a large publicly available dataset of chest radiographs with free-text radiology reports. Contains 377,110 images corresponding to 227,835 radiographic studies from 65,379 patients at Beth Israel Deaconess Medical Center. The dataset is de-identified and widely used for medical imaging research, automated report generation, and medical AI development.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Multimodal","Healthcare","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mimic-cxr","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8fad0036a02a982d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mimic-cxr"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mimic-cxr","url":"https://llm-stats.com/benchmarks/mimic-cxr","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","healthcare","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_82c06ed0fe65d988","familyId":"catalog_family_82c06ed0fe65d988","name":"MiMo Coding Bench","oneLine":"MiMo Coding Bench evaluates coding-agent capabilities on software engineering tasks reported with the MiMo model family.","description":"MiMo Coding Bench evaluates coding-agent capabilities on software engineering tasks reported with the MiMo model family.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mimo-coding-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_82c06ed0fe65d988"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mimo-coding-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mimo-coding-bench","url":"https://llm-stats.com/benchmarks/mimo-coding-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_mindedit-bench_02b6e953","familyId":"bmf_ebce6cd9eb55","name":"MindEdit-Bench","oneLine":"MindEdit-Bench evaluates counterfactual spatial reasoning in VLMs with 1,003 multiple-choice questions from private indoor scenes, covering six task types.","area":"Vision & 3D","applicationDomains":["Consumer & Productivity"],"primaryDomain":"Consumer & Productivity","industrySectors":["Consumer Technology"],"capabilities":["Reasoning","Geometric reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.00491","pdf":"https://arxiv.org/pdf/2607.00491","project":null,"code":null,"data":"https://huggingface.co/datasets/ZODAOfficial/MindEdit-Bench","hfPaper":"https://huggingface.co/papers/2607.00491"},"evidence":{"snippet":"We introduce MindEdit-Bench, a benchmark of six spatial reasoning tasks built from three-photo smartphone triplets of newly captured indoor scenes via an automatic in-the-wild 3D scene-graph extraction pipeline.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":150,"hfDatasetLikes":1},"source":{"type":"arxiv","id":"2607.00491"},"ranking":{"90d":{"score":40,"rank":153,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":43,"datasetRankPopulation":66}},"description":"MindEdit-Bench evaluates counterfactual spatial reasoning in VLMs with 1,003 multiple-choice questions from private indoor scenes, covering six task types.","whyItMatters":"Tests whether VLMs can reason about hypothetical object manipulations, a capability not covered by existing benchmarks, with human-verified answers.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d54d6d1a7d263f5a4b25aa43486eee2c5b6fccc75f219a1ad4181db3e06a7e67"},"motivation":"Benchmarks for vision-language models (VLMs) mostly test observational spatial reasoning: models describe relations already visible in the input.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.00491","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ZODAOfficial","organizationType":"community","sourceUrl":"https://huggingface.co/datasets/ZODAOfficial/MindEdit-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_69ebd5b4d828836c","familyId":"catalog_family_69ebd5b4d828836c","name":"Minerva","oneLine":"Minerva is a benchmark for complex video reasoning, evaluating models on multi-step reasoning over long and information-dense video content.","description":"Minerva is a benchmark for complex video reasoning, evaluating models on multi-step reasoning over long and information-dense video content.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/minerva","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_69ebd5b4d828836c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/minerva"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"minerva","url":"https://llm-stats.com/benchmarks/minerva","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","video","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_minexplore_36c767cc","familyId":"bmf_9cc4f14cd3f7","name":"MineXplore","oneLine":"MineXplore is an open-source MuJoCo-based benchmark for robot navigation in GNSS-denied underground mine environments, derived from the Leung et al. 2017 Chilean copper mine dataset. It reconstructs a 104,423 sq.m tunnel network with octagonal wall cross-sections, LiDAR-sourced jagged wall geometry, three terrain friction zones, a global 5 degree incline, and periodic spot lighting. The benchmark includes an evaluation protocol based on coverage percentage, with a 90% target for single-agent PPO policies.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.RO"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04569","pdf":"https://arxiv.org/pdf/2606.04569","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04569"},"evidence":{"snippet":"We present MineXplore, an open-source MuJoCo-based navigation benchmark derived from the Leung et al.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04569"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MineXplore is an open-source MuJoCo-based benchmark for robot navigation in GNSS-denied underground mine environments, derived from the Leung et al. 2017 Chilean copper mine dataset. It reconstructs a 104,423 sq.m tunnel network with octagonal wall cross-sections, LiDAR-sourced jagged wall geometry, three terrain friction zones, a global 5 degree incline, and periodic spot lighting. The benchmark includes an evaluation protocol based on coverage percentage, with a 90% target for single-agent PPO policies.","whyItMatters":"MineXplore addresses the lack of realistic simulation benchmarks for underground mine navigation in the open-source ecosystem, providing a GPU-compatible environment grounded in real production-mine geometry. It enables reproducible evaluation of reinforcement learning policies under degraded sensing and complex topology, supporting progress in autonomous navigation for safety-critical applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ab55c2f828f4ad663110534b49ad4afa591ab85e7c7a328164f43433350a8dcf"},"motivation":"Underground mines present extreme conditions for autonomous robot navigation: GPS is denied, lighting is degraded, and tunnel topology is loop-rich and non-convex.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04569","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_mira-ev_98696803","familyId":"bmf_29781ac889f4","name":"MIRA-Ev","oneLine":"MIRA-Ev evaluates clinical NLP on evidence detection and relational reasoning in Spanish MIR exam cases, with three tasks: evidence sentence retrieval, argumentative component extraction, and relation classification.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19201","pdf":"https://arxiv.org/pdf/2607.19201","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19201"},"evidence":{"snippet":"We introduce MIRA-Ev, a clinical argument mining benchmark built on Spanish M\\'edico Interno Residente (MIR) licensing-exam cases, re-annotated by expert clinicians with span-level premises, claims, and directed support/attack relations, and released in parallel Spanish (native), English, and Basque versions, the first clinical argumentation resource in Basque.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19201"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MIRA-Ev evaluates clinical NLP on evidence detection and relational reasoning in Spanish MIR exam cases, with three tasks: evidence sentence retrieval, argumentative component extraction, and relation classification.","whyItMatters":"It addresses the lack of evidence-level evaluation in clinical NLP, providing a multilingual benchmark (Spanish, English, Basque) to assess not just final answers but the grounding of diagnoses.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c0a11c9dcd59b70e231f8a97cab34962be656fa22530b8fa557992f07c849f6d"},"motivation":"Clinical NLP evaluation remains dominated by multiple-choice question answering (MCQA), which scores only final-answer accuracy and cannot detect when a model reaches the correct diagnosis while grounding it in irrelevant, absent, or contradictory evidence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19201","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_mira-math_da1f22ae","familyId":"bmf_543fc16ad0c8","name":"MIRA-Math","oneLine":"MIRA-Math evaluates mathematical reasoning where each problem is missing exactly one necessary atomic fact that must be requested in natural language under a strict budget, then integrated into an exact answer. It contains 2,310 instances across 22 typed mathematical families.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.07391","pdf":"https://arxiv.org/pdf/2607.07391","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.07391"},"evidence":{"snippet":"We introduce MIRA-Math, a benchmark for a narrower diagnostic capability: solving mathematical problems whose full latent state has a unique answer, but whose solver-facing view is missing exactly one necessary atomic fact.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.07391"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MIRA-Math evaluates mathematical reasoning where each problem is missing exactly one necessary atomic fact that must be requested in natural language under a strict budget, then integrated into an exact answer. It contains 2,310 instances across 22 typed mathematical families.","whyItMatters":"Isolates the diagnostic capability of minimal information requesting separate from broader tool use or long-horizon dialogue, revealing that request success and final-answer accuracy are separable.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d5082b9d9e03d5bd261cbd50436766fa6cb04e245b67357bddc3105979bc4ad1"},"motivation":"Mathematical reasoning benchmarks typically provide all facts needed to solve each problem, while interactive benchmarks often mix reasoning with tools, retrieval, and long-horizon dialogue.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.07391","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_mirabench_80851376","familyId":"bmf_46ceeefc922f","name":"MiraBench","oneLine":"MiraBench evaluates action-conditioned reliability in robotic world models through three levels: Physics Adherence, Action-Following Fidelity, and Optimism Bias Detection, using a human-annotated corpus of over 16,000 judgments.","area":"Language & Knowledge","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29360","pdf":"https://arxiv.org/pdf/2605.29360","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29360"},"evidence":{"snippet":"We introduce \\textsc{MiraBench}, a hierarchical benchmark that defines \\emph{action-conditioned reliability} as a core evaluation target for robotic world models.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29360"},"ranking":{},"description":"MiraBench evaluates action-conditioned reliability in robotic world models through three levels: Physics Adherence, Action-Following Fidelity, and Optimism Bias Detection, using a human-annotated corpus of over 16,000 judgments.","whyItMatters":"Existing benchmarks focus on visual fidelity, leaving unclear whether predicted futures are physically plausible, faithful to actions, and calibrated to failure. MiraBench aims to provide a diagnostic foundation for assessing world models as simulators, but its public evaluation protocol and release details are not yet specified.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6dc4925abdaa7ae18e4c4cd4f0d0499171b6bf8baddf6dac5dcdc0439e8bbf69"},"motivation":"Action-conditioned world models are increasingly used as scalable simulators for robot learning, yet current evaluations provide limited evidence that their predictions are reliable under the actions they condition on.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29360","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_mirrorcode_114f6eb8","familyId":"bmf_5f4db04ac40d","name":"MirrorCode","oneLine":"MirrorCode evaluates AI agents on reimplementing entire software projects from behavior only, matching outputs on end-to-end tests across 25 programs spanning Unix utilities, data serialization, bioinformatics, interpreters, static analysis, cryptography, and compression.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.30182","pdf":"https://arxiv.org/pdf/2606.30182","project":null,"code":"https://github.com/epoch-research/MirrorCode","data":null,"hfPaper":"https://huggingface.co/papers/2606.30182"},"evidence":{"snippet":"To address these challenges, we introduce MirrorCode, a long-horizon coding benchmark based on reimplementing entire software projects.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":70,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.30182"},"ranking":{"90d":{"score":54,"rank":38,"coverage":0.55,"confidence":"Low"}},"description":"MirrorCode evaluates AI agents on reimplementing entire software projects from behavior only, matching outputs on end-to-end tests across 25 programs spanning Unix utilities, data serialization, bioinformatics, interpreters, static analysis, cryptography, and compression.","whyItMatters":"Existing coding benchmarks focus on shorter tasks, while long-horizon reimplementation remains hard to compare. MirrorCode provides a standardized, repeatable measure of autonomous software engineering capability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"67c61e915e20b607b728481872859b99bae0e3b00f5681543f13e73b2ec5d59e"},"motivation":"AI models are rapidly improving at autonomous coding, as shown by benchmark progress and one-off demonstrations such as AI implementing a C compiler.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.30182","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Epoch Research","organizationType":"academic-lab","sourceUrl":"https://github.com/epoch-research/MirrorCode","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_mirrorcraft_4e4ff3e9","familyId":"bmf_17f3ccda834b","name":"MirrorCraft","oneLine":"MirrorCraft is a paired benchmark for evaluating LLM-based agents under hidden rule changes in Minecraft. Each Mirror world copies a Vanilla world with modified server-side rules, while terrain and objectives match. It includes five biomes, six rule suites, three objectives, and uses deterministic milestones and Rule Intervention Effect.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.29218","pdf":"https://arxiv.org/pdf/2607.29218","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.29218"},"evidence":{"snippet":"In this paper, we introduce MirrorCraft, a paired benchmark for evaluating agents under hidden rule changes in Minecraft.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.29218"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MirrorCraft is a paired benchmark for evaluating LLM-based agents under hidden rule changes in Minecraft. Each Mirror world copies a Vanilla world with modified server-side rules, while terrain and objectives match. It includes five biomes, six rule suites, three objectives, and uses deterministic milestones and Rule Intervention Effect.","whyItMatters":"Most Minecraft benchmarks use fixed mechanics, not testing adaptability to rule changes. MirrorCraft provides a controlled setting to study agent performance under hidden rule modifications, important for developing agents that can handle dynamic environments.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"047b7389d34807385728a55dbff7e5866c34b714d09e80dc54496f6f82f01002"},"motivation":"With the prosperity of the large language models (LLMs), it has become an interesting topic: how do LLM-based agents work in Minecraft?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.29218","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_missingbench-verified_db33d9c0","familyId":"bmf_d84428e12b3d","name":"MissingBench-Verified","oneLine":"MissingBench-Verified evaluates vision-language models on detecting missing object parts in images, covering ten leading models with consistent failure rates and probing mitigation strategies.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.18673","pdf":"https://arxiv.org/pdf/2607.18673","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.18673"},"evidence":{"snippet":"We present MissingBench-Verified, a benchmark designed to evaluate a specific and practically relevant scenario: when vision-language models fail to recognize that an essential component of an object has been removed.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.18673"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MissingBench-Verified evaluates vision-language models on detecting missing object parts in images, covering ten leading models with consistent failure rates and probing mitigation strategies.","whyItMatters":"It highlights a fundamental limitation of VLMs for inspection tasks, showing that current prompting and post-hoc corrections are insufficient, which informs the need for architectural changes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"50ee98a9487b2fa81dab8a16cd28b7ce977e62d1839a8ff1543b23e60e7ce3c5"},"motivation":"Vision Language Models (VLMs) are well known for hallucinating non-existent objects in images.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.18673","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_missionbench_cc95e3a2","familyId":"bmf_a4dfacf50588","name":"MissionBench","oneLine":"A benchmark for mission-level evaluation of MLLMs in aerial 3D environments, consisting of 120 missions across five simulated environments and four task families.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Geometric reasoning"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.22014","pdf":"https://arxiv.org/pdf/2607.22014","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.22014"},"evidence":{"snippet":"We introduce MissionBench, a benchmark for mission-level evaluation of MLLMs in aerial 3D environments.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.22014"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark for mission-level evaluation of MLLMs in aerial 3D environments, consisting of 120 missions across five simulated environments and four task families.","whyItMatters":"Highlights the difficulty of zero-shot long-horizon embodied tasks and the need for closed-loop evaluation, motivating scaling-driven improvements.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0f2dbb9535534373f1396007bf234da9f9e3d2e467a1c29ae7ffaa72d8c2bfdb"},"motivation":"Multimodal Large Language Models (MLLMs) are emerging as core reasoning modules for embodied agents, yet it remains unclear how well general-purpose models can solve long-horizon embodied tasks from a single high-level instruction.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.22014","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_mkg-rag-bench_ba53e1c9","familyId":"bmf_7f5057aa0709","name":"MKG-RAG-Bench","oneLine":"MKG-RAG-Bench is a benchmark for retrieval in multimodal knowledge graph-augmented generation, constructed from two knowledge graphs (general and medical) with QA datasets supporting controlled evaluation of retrieval and generation. It uses an LLM-based curation pipeline for structurally grounded queries.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval","Factuality"],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.26458","pdf":"https://arxiv.org/pdf/2606.26458","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.26458"},"evidence":{"snippet":"To address this gap, we introduce MKG-RAG-Bench, a cross-domain benchmark explicitly designed to evaluate retrieval in MKG-RAG.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26458"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MKG-RAG-Bench is a benchmark for retrieval in multimodal knowledge graph-augmented generation, constructed from two knowledge graphs (general and medical) with QA datasets supporting controlled evaluation of retrieval and generation. It uses an LLM-based curation pipeline for structurally grounded queries.","whyItMatters":"Isolates retrieval as a first-class evaluation target in MKG-RAG, addressing the challenge of heterogeneous multimodal knowledge. It provides a foundation for diagnosing retrieval limitations and improving end-to-end RAG systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4b4f1f01a0b0f9f02a4ee2efaa16e948f1c361424466ab402743574d523afae4"},"motivation":"Retrieval-augmented generation (RAG) over knowledge graphs has emerged as a promising approach for grounding large language models, yet existing benchmarks largely overlook the challenges of retrieval in multimodal knowledge graph RAG (MKG-RAG).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"KDD'26","evidence":"Accepted by KDD'26","evidenceUrl":"https://arxiv.org/abs/2606.26458","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"KDD'26","reviewStatus":"accepted","decisionRaw":"Accepted by KDD'26","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.26458","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by KDD'26","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_e34928c4f86c04c1","familyId":"catalog_family_e34928c4f86c04c1","name":"MKQA-11","oneLine":"A display-only multilingual QA retrieval benchmark reported by Liquid AI for LFM2.5 retriever models, using Recall@20 across 11 languages.","description":"A display-only multilingual QA retrieval benchmark reported by Liquid AI for LFM2.5 retriever models, using Recall@20 across 11 languages.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multilingual"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.liquid.ai/blog/lfm2-5-retrievers","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e34928c4f86c04c1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mkqa11"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mkqa11","url":"https://benchlm.ai/benchmarks/mkqa11","paperUrl":"https://www.liquid.ai/blog/lfm2-5-retrievers","year":"2026","fullName":"MKQA-11 multilingual retrieval","format":"Recall@20 average","tasks":"Cross-lingual open-domain QA retrieval","successorKey":null}],"catalogCategories":["multilingual"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"lib_mle_bench","familyId":"family_mle_bench","name":"MLE-bench","oneLine":"Established benchmark family · Agents & Tool Use.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents & Tool Use"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2410.07095","pdf":null,"project":"https://github.com/openai/mle-bench","code":"https://github.com/openai/mle-bench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_mle_bench"},"ranking":{},"recordType":"family","aliases":["MLE Bench"],"sourceAttribution":[{"role":"official-repository","url":"https://github.com/openai/mle-bench"}],"adoptionRefs":["openai-gpt5"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Agents","Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"mle-bench","url":"https://llm-stats.com/benchmarks/mle-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":2,"catalogStarCount":0},{"id":"catalog_eb8fe21cec35acc2","familyId":"catalog_family_eb8fe21cec35acc2","name":"MLE-Bench Lite","oneLine":"MLE-Bench Lite evaluates AI agents on machine learning engineering tasks, testing their ability to build, train, and optimize ML models for Kaggle-style competitions in a lightweight evaluation format.","description":"MLE-Bench Lite evaluates AI agents on machine learning engineering tasks, testing their ability to build, train, and optimize ML models for Kaggle-style competitions in a lightweight evaluation format.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.minimax.io/news/minimax-m27-en","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_eb8fe21cec35acc2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mle-bench-lite"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mle-bench-lite"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mleBenchLite","url":"https://benchlm.ai/benchmarks/mle-bench-lite","paperUrl":"https://www.minimax.io/news/minimax-m27-en","year":"2026","fullName":"MLE-Bench Lite","format":"Autonomous iterative ML optimization","tasks":"Low-resource ML competitions","successorKey":null},{"catalog":"llm-stats","sourceId":"mle-bench-lite","url":"https://llm-stats.com/benchmarks/mle-bench-lite","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_mlingualfc_4a4cc033","familyId":"bmf_aee0e158e17e","name":"MLingualFC","oneLine":"MLingualFC evaluates jailbreak vulnerabilities in multilingual vision-language models using flowchart images encoding harmful instructions in five languages, measuring attack success rates.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07706","pdf":"https://arxiv.org/pdf/2606.07706","project":null,"code":"https://github.com/Rishabhpm23/MLingualFC","data":null,"hfPaper":"https://huggingface.co/papers/2606.07706"},"evidence":{"snippet":"In this paper, we introduce MLingualFC, a multilingual multimodal benchmark designed to evaluate jailbreak vulnerabilities of VLMs across diverse languages using structured flowchart representations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07706"},"ranking":{"90d":{"score":28,"rank":284,"coverage":0.55,"confidence":"Low"}},"description":"MLingualFC evaluates jailbreak vulnerabilities in multilingual vision-language models using flowchart images encoding harmful instructions in five languages, measuring attack success rates.","whyItMatters":"Safety alignment in multilingual VLMs is under-tested; this benchmark highlights gaps across languages, aiding safety evaluations, but lacks a standalone public comparison path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6d85fccf1244eeebc796f9afc5367050b859c14c5e20eac01af5ad98c8c5de26"},"motivation":"Vision-Language Models (VLMs) have demonstrated strong performance across multimodal tasks, yet their safety robustness remains an open challenge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07706","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Rishabhpm23","organizationType":"community","sourceUrl":"https://github.com/Rishabhpm23/MLingualFC","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_413ce4ccc028faaf","familyId":"catalog_family_413ce4ccc028faaf","name":"MLS-Bench Lite","oneLine":"MLS-Bench Lite is the official 30-task subset of MLS-Bench for evaluating whether AI systems can invent generalizable and scalable machine learning methods across LLM pretraining and post-training, robotics, world models, computer vision, reinforcement learning, optimization, ML systems, and AI for Science.","description":"MLS-Bench Lite is the official 30-task subset of MLS-Bench for evaluating whether AI systems can invent generalizable and scalable machine learning methods across LLM pretraining and post-training, robotics, world models, computer vision, reinforcement learning, optimization, ML systems, and AI for Science.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Reasoning","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://mls-bench.com/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_413ce4ccc028faaf"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mlsbenchlite"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mls-bench-lite"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mlsBenchLite","url":"https://benchlm.ai/benchmarks/mlsbenchlite","paperUrl":"https://mls-bench.com/","year":"2026","fullName":"MLS-Bench Lite","format":"Agentic ML task evaluation","tasks":"30 machine-learning research tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"mls-bench-lite","url":"https://llm-stats.com/benchmarks/mls-bench-lite","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","reasoning","agents","code"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_mlubench_e20fa573","familyId":"bmf_8d9e10142c64","name":"MLUBench","oneLine":"MLUBench evaluates multimodal large language models on lifelong unlearning across 127 entities and 9 classes, providing QA pairs and images. It includes a protocol for sequential unlearning requests and evaluation of forgetting and retention.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.12809","pdf":"https://arxiv.org/pdf/2606.12809","project":null,"code":"https://github.com/lihe-maxsize/Lifelong_Unlearning_main","data":null,"hfPaper":"https://huggingface.co/papers/2606.12809"},"evidence":{"snippet":"To fill this gap, we introduce the MLUBench, a large-scale and comprehensive benchmark featuring 127 entities across 9 classes under lifelong unlearning requests.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.12809"},"ranking":{"90d":{"score":23,"rank":307,"coverage":0.7,"confidence":"Medium"}},"description":"MLUBench evaluates multimodal large language models on lifelong unlearning across 127 entities and 9 classes, providing QA pairs and images. It includes a protocol for sequential unlearning requests and evaluation of forgetting and retention.","whyItMatters":"Lifelong unlearning is a practical challenge for MLLMs as data removal requests arrive over time. This benchmark enables systematic evaluation of unlearning methods and highlights the unique constraint of preserving multimodal alignment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"02c0e16073f9c2b6fad8fb9dd7b135c9ffd55336756967d2eb09fd92d89be05d"},"motivation":"Multimodal large language models (MLLMs) are trained on massive multimodal data, making data unlearning increasingly important as data owners may request the removal of specific content.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026","evidence":"36 pages, accepted to the ICML 2026","evidenceUrl":"https://arxiv.org/abs/2606.12809","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026","reviewStatus":"accepted","decisionRaw":"36 pages, accepted to the ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.12809","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"36 pages, accepted to the ICML 2026","level":"author-claim"}]}],"publishers":[{"name":"MLUBench Project","organizationType":"community","sourceUrl":"https://github.com/lihe-maxsize/Lifelong_Unlearning_main","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_737363d33f94b15d","familyId":"catalog_family_737363d33f94b15d","name":"MLVU","oneLine":"A comprehensive benchmark for multi-task long video understanding that evaluates multimodal large language models on videos ranging from 3 minutes to 2 hours across 9 distinct tasks including reasoning, captioning, recognition, and summarization.","description":"A comprehensive benchmark for multi-task long video understanding that evaluates multimodal large language models on videos ranging from 3 minutes to 2 hours across 9 distinct tasks including reasoning, captioning, recognition, and summarization.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Multimodal","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mlvu","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_737363d33f94b15d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mlvu"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mlvu","url":"https://llm-stats.com/benchmarks/mlvu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","multimodal","video","vision"],"catalogModelCount":10,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_5a1b77f6f2eb5c8c","familyId":"catalog_family_5a1b77f6f2eb5c8c","name":"MLVU (M-Avg)","oneLine":"A multi-task video understanding benchmark averaged across MLVU categories.","description":"A multi-task video understanding benchmark averaged across MLVU categories.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5a1b77f6f2eb5c8c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mlvuavg"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mlvuAvg","url":"https://benchlm.ai/benchmarks/mlvuavg","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"MLVU mean average","format":"Video QA and understanding","tasks":"General video understanding","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_533dd059f93116c0","familyId":"catalog_family_533dd059f93116c0","name":"MLVU-M","oneLine":"MLVU-M benchmark","description":"MLVU-M benchmark","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mlvu-m","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_533dd059f93116c0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mlvu-m"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mlvu-m","url":"https://llm-stats.com/benchmarks/mlvu-m","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["general"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_da559c403a907e7f","familyId":"catalog_family_da559c403a907e7f","name":"MM IF-Eval","oneLine":"A challenging multimodal instruction-following benchmark that includes both compose-level constraints for output responses and perception-level constraints tied to input images, with comprehensive evaluation pipeline.","description":"A challenging multimodal instruction-following benchmark that includes both compose-level constraints for output responses and perception-level constraints tied to input images, with comprehensive evaluation pipeline.","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Structured Output"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mm-if-eval","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_da559c403a907e7f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mm-if-eval"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mm-if-eval","url":"https://llm-stats.com/benchmarks/mm-if-eval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","structured output"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general"},{"id":"catalog_1f4900925fcf51d0","familyId":"catalog_family_1f4900925fcf51d0","name":"MM-BrowserComp","oneLine":"MM-BrowserComp evaluates multimodal agents on web browsing and information retrieval tasks, testing a model's ability to perceive, navigate, and extract information from real web environments.","description":"MM-BrowserComp evaluates multimodal agents on web browsing and information retrieval tasks, testing a model's ability to perceive, navigate, and extract information from real web environments.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Search","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mm-browsercomp","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1f4900925fcf51d0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mm-browsercomp"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mm-browsercomp","url":"https://llm-stats.com/benchmarks/mm-browsercomp","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","search","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_a46d809efa147169","familyId":"catalog_family_a46d809efa147169","name":"MM-ClawBench","oneLine":"MM-ClawBench evaluates models on MiniMax's Claw-style agent benchmark, measuring practical agentic task completion quality in real-world OpenClaw usage scenarios.","description":"MM-ClawBench evaluates models on MiniMax's Claw-style agent benchmark, measuring practical agentic task completion quality in real-world OpenClaw usage scenarios.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.minimax.io/news/minimax-m27-en","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a46d809efa147169"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mmclawbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mm-clawbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mmClawBench","url":"https://benchlm.ai/benchmarks/mmclawbench","paperUrl":"https://www.minimax.io/news/minimax-m27-en","year":"2026","fullName":"MM-ClawBench","format":"Agent workflow evaluation","tasks":"OpenClaw-style real-world tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"mm-clawbench","url":"https://llm-stats.com/benchmarks/mm-clawbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_mm-creativitybench_e366f23e","familyId":"bmf_534ca0e1c4b9","name":"MM-CreativityBench","oneLine":"MM-CreativityBench evaluates large multimodal models on affordance-grounded creative tool use, requiring scene inspection, entity/part selection, and physically feasible solutions in visually rich environments.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agents","Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26396","pdf":"https://arxiv.org/pdf/2605.26396","project":null,"code":"https://github.com/CreativityBench/MM-CreativityBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.26396"},"evidence":{"snippet":"To evaluate this ability, we introduce MM-CreativityBench, a benchmark for affordance-grounded creative tool use in visually rich, physically constrained environments.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":21,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26396"},"ranking":{},"description":"MM-CreativityBench evaluates large multimodal models on affordance-grounded creative tool use, requiring scene inspection, entity/part selection, and physically feasible solutions in visually rich environments.","whyItMatters":"It probes beyond pattern recognition, assessing grounded exploration and reasoning. Current models show gaps in exploration and hallucination, motivating preference-based alignment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d54abdaedabbfb3704140577574bd4b6c6df625f528f74f7076565e78c9f6678"},"motivation":"Large multimodal models (LMMs) have rapidly advanced in perception and reasoning; however, it remains unclear whether these capabilities generalize to discovering visually grounded solutions in open-ended environments, beyond pattern recognition.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26396","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MM-CreativityBench team","organizationType":"academic-lab","sourceUrl":"https://github.com/CreativityBench/MM-CreativityBench","role":"benchmark-publisher"}],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_37a7875647e70c76","familyId":"catalog_family_37a7875647e70c76","name":"MM-Mind2Web","oneLine":"A multimodal web navigation benchmark comprising 2,000 open-ended tasks spanning 137 websites across 31 domains. Each task includes HTML documents paired with webpage screenshots, action sequences, and complex web interactions.","description":"A multimodal web navigation benchmark comprising 2,000 open-ended tasks spanning 137 websites across 31 domains. Each task includes HTML documents paired with webpage screenshots, action sequences, and complex web interactions.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Frontend Development","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mm-mind2web","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_37a7875647e70c76"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mm-mind2web"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mm-mind2web","url":"https://llm-stats.com/benchmarks/mm-mind2web","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","frontend development","agents"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_1165a051eddbecad","familyId":"catalog_family_1165a051eddbecad","name":"MM-MT-Bench","oneLine":"A multi-turn LLM-as-a-judge evaluation benchmark for testing multimodal instruction-tuned models' ability to follow user instructions in multi-turn dialogues and answer open-ended questions in a zero-shot manner.","description":"A multi-turn LLM-as-a-judge evaluation benchmark for testing multimodal instruction-tuned models' ability to follow user instructions in multi-turn dialogues and answer open-ended questions in a zero-shot manner.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Communication"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mm-mt-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1165a051eddbecad"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mm-mt-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mm-mt-bench","url":"https://llm-stats.com/benchmarks/mm-mt-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","communication"],"catalogModelCount":17,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mm-snowball_ce633f35","familyId":"bmf_17f078aecdcc","name":"MM-Snowball","oneLine":"MM-Snowball is a benchmark for diagnosing hallucination snowballing in multimodal multi-turn dialogue, with fine-grained analysis. It includes data and code via a project page.","area":"Multimodal","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00622","pdf":"https://arxiv.org/pdf/2606.00622","project":"https://frenkie-chiang.github.io/MM-Snowball","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.00622"},"evidence":{"snippet":"To address this, we introduce MM-Snowball, the first benchmark for fine-grained diagnosis of hallucination snowballing within dialogues.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00622"},"ranking":{},"description":"MM-Snowball is a benchmark for diagnosing hallucination snowballing in multimodal multi-turn dialogue, with fine-grained analysis. It includes data and code via a project page.","whyItMatters":"Addresses the lack of benchmarks for error propagation in long-horizon interactions, and shows existing mitigation methods are ineffective, motivating new approaches.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"34a3475b761d9810c0eb1b66e2adde6c1559b2de348d06a491b1b222eed3cdfe"},"motivation":"Multimodal large language models (MLLMs) demonstrate remarkable visual understanding, yet their reliability in interactive settings is severely undermined by hallucination snowballing: a phenomenon where initial errors amplify across conversational turns, leading to a collapse in coherence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"International Conference on Machine Learning (ICML 2026)","evidence":"Accepted by The International Conference on Machine Learning (ICML 2026)","evidenceUrl":"https://arxiv.org/abs/2606.00622","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"International Conference on Machine Learning (ICML 2026)","reviewStatus":"accepted","decisionRaw":"Accepted by The International Conference on Machine Learning (ICML 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.00622","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by The International Conference on Machine Learning (ICML 2026)","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_mmae_ba12849c","familyId":"bmf_ba3363208910","name":"MMAE","oneLine":"MMAE is a benchmark for instruction-based audio editing with 2,000 samples across 7 modalities, 6 complexity levels, and rubric-based evaluation with 17,741 criteria for instruction following and context consistency.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SD"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07229","pdf":"https://arxiv.org/pdf/2606.07229","project":null,"code":"https://github.com/ddlBoJack/MMAE","data":null,"hfPaper":"https://huggingface.co/papers/2606.07229"},"evidence":{"snippet":"We introduce MMAE, a Massive Multitask Audio Editing benchmark, serving as the first comprehensive evaluation testbed designed for general-purpose instruction-based audio editing.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":45,"hfDailySubmittedAt":"2026-06-08T00:00:00.000Z","githubStars":103,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07229"},"ranking":{"90d":{"score":60,"rank":18,"coverage":0.7,"confidence":"Medium"}},"description":"MMAE is a benchmark for instruction-based audio editing with 2,000 samples across 7 modalities, 6 complexity levels, and rubric-based evaluation with 17,741 criteria for instruction following and context consistency.","whyItMatters":"It provides the first comprehensive evaluation testbed for general-purpose audio editing, enabling precise multi-dimensional assessment and identifying bottlenecks in current models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4e533933f8a78d6ec2a7735826001bb55f53e2cb74f909f504a2f9b467d6184c"},"motivation":"We introduce MMAE, a Massive Multitask Audio Editing benchmark, serving as the first comprehensive evaluation testbed designed for general-purpose instruction-based audio editing.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07229","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_484ddc4711d4f077","familyId":"catalog_family_484ddc4711d4f077","name":"MMAnswerBench","oneLine":"A multimodal mathematical reasoning benchmark that tests whether models can answer visually grounded math questions correctly.","description":"A multimodal mathematical reasoning benchmark that tests whether models can answer visually grounded math questions correctly.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_484ddc4711d4f077"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mmanswerbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mmAnswerBench","url":"https://benchlm.ai/benchmarks/mmanswerbench","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"MMAnswerBench","format":"Visual and structured mathematical QA","tasks":"Multimodal math questions","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_mmarch_3c23535d","familyId":"bmf_cde8c1453aed","name":"MMArch","oneLine":"MMArch is a benchmark for multimodal reasoning in architecture and civil engineering, spanning ten subdomains and built from figures in peer-reviewed papers. It contains 1,212 short-answer items requiring perception, principle identification, and application.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09281","pdf":"https://arxiv.org/pdf/2608.09281","project":"https://dcx-swjtu.github.io/MMArch/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09281"},"evidence":{"snippet":"We introduce MMArch, a benchmark for architecture and civil engineering spanning ten subdomains and built entirely from figures in peer-reviewed papers.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09281"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MMArch is a benchmark for multimodal reasoning in architecture and civil engineering, spanning ten subdomains and built from figures in peer-reviewed papers. It contains 1,212 short-answer items requiring perception, principle identification, and application.","whyItMatters":"Existing benchmarks test drawing recognition or information extraction, but MMArch evaluates whether models can combine distributed visual evidence with engineering principles, filling a gap in multimodal reasoning evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1791fc684d00be2eaa297e2ddec1284cf0fe1b3277486b536ab94ee02070430f"},"motivation":"Multimodal large language models (MLLMs) perform strongly on engineering imagery, yet existing benchmarks mostly test drawing recognition, information extraction, or compliance checking, leaving open whether models can combine distributed visual evidence with engineering principles to reach a conclusion.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09281","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"dcx-swjtu","organizationType":"academic-lab","sourceUrl":"https://dcx-swjtu.github.io/MMArch/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_58bca2a148fba12e","familyId":"catalog_family_58bca2a148fba12e","name":"MMAU","oneLine":"A massive multi-task audio understanding and reasoning benchmark comprising 10,000 carefully curated audio clips paired with human-annotated natural language questions spanning speech, environmental sounds, and music. Requires expert-level knowledge and complex reasoning across 27 distinct skills.","description":"A massive multi-task audio understanding and reasoning benchmark comprising 10,000 carefully curated audio clips paired with human-annotated natural language questions spanning speech, environmental sounds, and music. Requires expert-level knowledge and complex reasoning across 27 distinct skills.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmau","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_58bca2a148fba12e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmau"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmau","url":"https://llm-stats.com/benchmarks/mmau","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","audio"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_a5b7b90c3d77b196","familyId":"catalog_family_a5b7b90c3d77b196","name":"MMAU Music","oneLine":"A subset of the MMAU benchmark focused specifically on music understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across music audio clips.","description":"A subset of the MMAU benchmark focused specifically on music understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across music audio clips.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmau-music","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a5b7b90c3d77b196"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmau-music"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmau-music","url":"https://llm-stats.com/benchmarks/mmau-music","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","audio"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_ae6c7417fddf3e80","familyId":"catalog_family_ae6c7417fddf3e80","name":"MMAU Sound","oneLine":"A subset of the MMAU benchmark focused specifically on environmental sound understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across environmental sound clips.","description":"A subset of the MMAU benchmark focused specifically on environmental sound understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across environmental sound clips.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmau-sound","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ae6c7417fddf3e80"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmau-sound"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmau-sound","url":"https://llm-stats.com/benchmarks/mmau-sound","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","audio"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_b62294e604454e72","familyId":"catalog_family_b62294e604454e72","name":"MMAU Speech","oneLine":"A subset of the MMAU benchmark focused specifically on speech understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across speech audio clips.","description":"A subset of the MMAU benchmark focused specifically on speech understanding and reasoning tasks. Part of a comprehensive multimodal audio understanding benchmark that evaluates models on expert-level knowledge and complex reasoning across speech audio clips.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Speech To Text","Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmau-speech","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b62294e604454e72"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmau-speech"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmau-speech","url":"https://llm-stats.com/benchmarks/mmau-speech","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","speech to text","audio"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_02e126e011402c43","familyId":"catalog_family_02e126e011402c43","name":"MMBC","oneLine":"MMBC is a multimodal benchmark for vision-language knowledge and reasoning.","description":"MMBC is a multimodal benchmark for vision-language knowledge and reasoning.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmbc","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_02e126e011402c43"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmbc"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmbc","url":"https://llm-stats.com/benchmarks/mmbc","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","multimodal","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_26ec0b897cf3f983","familyId":"catalog_family_26ec0b897cf3f983","name":"MMBench","oneLine":"A bilingual benchmark for assessing multi-modal capabilities of vision-language models through multiple-choice questions in both English and Chinese, providing systematic evaluation across diverse vision-language tasks with robust metrics.","description":"A bilingual benchmark for assessing multi-modal capabilities of vision-language models through multiple-choice questions in both English and Chinese, providing systematic evaluation across diverse vision-language tasks with robust metrics.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_26ec0b897cf3f983"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmbench","url":"https://llm-stats.com/benchmarks/mmbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":9,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mmbench-live_b26d79ee","familyId":"bmf_23818de44e7f","name":"MMBench-Live","oneLine":"MMBench-Live is a continuously evolving multimodal benchmark built by a multi-agent pipeline from MMBench. It contains 5.9K newly generated evaluation instances with executable reasoning, and evaluates vision-language models across question-answer generation and reasoning tasks. Updates cost about USD 30 and take 1-2 hours, with distribution-consistent strategy to maintain comparability.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.01813","pdf":"https://arxiv.org/pdf/2607.01813","project":null,"code":"https://github.com/PRIS-CV/MMBench-Live","data":null,"hfPaper":"https://huggingface.co/papers/2607.01813"},"evidence":{"snippet":"We present MMBench-Live, a continuously evolving multimodal benchmark built by a multi-agent-driven automated pipeline.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01813"},"ranking":{"90d":{"score":28,"rank":276,"coverage":0.55,"confidence":"Low"}},"description":"MMBench-Live is a continuously evolving multimodal benchmark built by a multi-agent pipeline from MMBench. It contains 5.9K newly generated evaluation instances with executable reasoning, and evaluates vision-language models across question-answer generation and reasoning tasks. Updates cost about USD 30 and take 1-2 hours, with distribution-consistent strategy to maintain comparability.","whyItMatters":"Static benchmarks suffer from contamination and staleness; this benchmark provides a scalable, low-cost paradigm for sustainable evaluation that preserves model rankings and reduces memorization signals.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e87f2bd6c4e4ba1d586557018806e145d3d6044c1d41880947541bb3481736c4"},"motivation":"Evaluation benchmarks are essential for assessing vision-language models (VLMs), but most multimodal benchmarks are static, making them vulnerable to temporal staleness, data contamination, and costly maintenance.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01813","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_1f8d5e87988f20b7","familyId":"catalog_family_1f8d5e87988f20b7","name":"MMBench-V1.1","oneLine":"Version 1.1 of MMBench, an improved bilingual benchmark for assessing multi-modal capabilities of vision-language models through multiple-choice questions in both English and Chinese, providing systematic evaluation across diverse vision-language tasks.","description":"Version 1.1 of MMBench, an improved bilingual benchmark for assessing multi-modal capabilities of vision-language models through multiple-choice questions in both English and Chinese, providing systematic evaluation across diverse vision-language tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmbench-v1.1","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1f8d5e87988f20b7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmbench-v1.1"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmbench-v1.1","url":"https://llm-stats.com/benchmarks/mmbench-v1.1","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":20,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_fffd1561ca889f23","familyId":"catalog_family_fffd1561ca889f23","name":"MMBench-Video","oneLine":"A long-form multi-shot benchmark for holistic video understanding that incorporates approximately 600 web videos from YouTube spanning 16 major categories, with each video ranging from 30 seconds to 6 minutes. Includes roughly 2,000 original question-answer pairs covering 26 fine-grained capabilities.","description":"A long-form multi-shot benchmark for holistic video understanding that incorporates approximately 600 web videos from YouTube spanning 16 major categories, with each video ranging from 30 seconds to 6 minutes. Includes roughly 2,000 original question-answer pairs covering 26 fine-grained capabilities.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmbench-video","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fffd1561ca889f23"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmbench-video"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmbench-video","url":"https://llm-stats.com/benchmarks/mmbench-video","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","video","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mmdg-bench_d6d5ab19","familyId":"bmf_403771f7adde","name":"MMDG-Bench","oneLine":"MMDG-Bench is a benchmark for multi-modal domain generalization, providing two frameworks (D2M and M2D) and unified protocols across action recognition and face anti-spoofing tasks. Includes code repository.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00891","pdf":"https://arxiv.org/pdf/2606.00891","project":null,"code":"https://github.com/qszhan/MMDG-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.00891"},"evidence":{"snippet":"To address this, we introduce MMDG-Bench, a comprehensive benchmark featuring two foundational frameworks: DG then MML (D2M) and MML then DG (M2D).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00891"},"ranking":{},"description":"MMDG-Bench is a benchmark for multi-modal domain generalization, providing two frameworks (D2M and M2D) and unified protocols across action recognition and face anti-spoofing tasks. Includes code repository.","whyItMatters":"Provides a principled foundation for MMDG research, with insights on framework choice and backbone effects, and offers design guidelines for multi-modal robustness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"69baafbd62366f6aa7e0e2e25a93a336c149636d2fd8d5d37633d4a5ff8ded9f"},"motivation":"Multi-modal Domain Generalization (MMDG) seeks to leverage complementary modalities to enhance model robustness on unseen domains.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00891","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_c3583a3c34e250f1","familyId":"catalog_family_c3583a3c34e250f1","name":"MME","oneLine":"A comprehensive evaluation benchmark for Multimodal Large Language Models measuring both perception and cognition abilities across 14 subtasks. Features manually designed instruction-answer pairs to avoid data leakage and provides systematic quantitative assessment of MLLM capabilities.","description":"A comprehensive evaluation benchmark for Multimodal Large Language Models measuring both perception and cognition abilities across 14 subtasks. Features manually designed instruction-answer pairs to avoid data leakage and provides systematic quantitative assessment of MLLM capabilities.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mme","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c3583a3c34e250f1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mme"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mme","url":"https://llm-stats.com/benchmarks/mme","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_0459b9ac661bb094","familyId":"catalog_family_0459b9ac661bb094","name":"MME-RealWorld","oneLine":"A comprehensive evaluation benchmark for Multimodal Large Language Models featuring over 13,366 high-resolution images and 29,429 question-answer pairs across 43 subtasks and 5 real-world scenarios. The largest manually annotated multimodal benchmark to date, designed to test MLLMs on challenging high-resolution real-world scenarios.","description":"A comprehensive evaluation benchmark for Multimodal Large Language Models featuring over 13,366 high-resolution images and 29,429 question-answer pairs across 43 subtasks and 5 real-world scenarios. The largest manually annotated multimodal benchmark to date, designed to test MLLMs on challenging high-resolution real-world scenarios.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","General","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mme-realworld","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0459b9ac661bb094"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mme-realworld"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mme-realworld","url":"https://llm-stats.com/benchmarks/mme-realworld","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","general","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mmed-bench-ir_04b38f2d","familyId":"bmf_00f594160510","name":"MMed-Bench-IR","oneLine":"MMed-Bench-IR evaluates multilingual medical information retrieval across 6 languages with three tasks: cross-lingual QA retrieval, concept discrimination, and evidence retrieval for RAG.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Information retrieval"],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24200","pdf":"https://arxiv.org/pdf/2606.24200","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.24200"},"evidence":{"snippet":"We introduce MMed-Bench-IR, a benchmark designed to disentangle these axes across 6 languages and three structurally heterogeneous tasks: (1) cross-lingual medical QA retrieval with 6,127 queries grounded in the Unified Medical Language System (UMLS), (2) concept discrimination over 4,975 confusion sets at three difficulty tiers, and (3) multilingual evidence retrieval for RAG with 2,040 quality-assured queries.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24200"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MMed-Bench-IR evaluates multilingual medical information retrieval across 6 languages with three tasks: cross-lingual QA retrieval, concept discrimination, and evidence retrieval for RAG.","whyItMatters":"Uncovers severe cross-lingual failures in biomedical encoders that English-only benchmarks miss, highlighting the need for multilingual capability measurement in clinical RAG.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1eccf7d331acf480dc1cb1010e929f438b076b6eb13cac6834138aa764c5ad35"},"motivation":"Retrieval-augmented generation (RAG) in clinical settings increasingly requires multilingual retrieval against predominantly English evidence corpora.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24200","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"specific"},{"id":"bm_mmgist_394485cc","familyId":"bmf_06d36f826ac3","name":"MMGist","oneLine":"MMGist is a curated multimodal benchmark covering seven capability dimensions with 7,262 items, derived from 18 existing benchmarks via filtering pipelines. It evaluates vision-language models across dimensions such as Visual Logic and Expert Knowledge.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22437","pdf":"https://arxiv.org/pdf/2606.22437","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.22437"},"evidence":{"snippet":"To this end, we propose MMGist, a curated benchmark that covers seven capability dimensions and contains 7,262 items.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22437"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MMGist is a curated multimodal benchmark covering seven capability dimensions with 7,262 items, derived from 18 existing benchmarks via filtering pipelines. It evaluates vision-language models across dimensions such as Visual Logic and Expert Knowledge.","whyItMatters":"MMGist addresses the need for reliable and discriminative evaluation of vision-language models, aiming to reduce redundancy and saturation in existing benchmarks while preserving model rankings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e4221c16868e4b94571867d898969f7c8b4c85170f3ea03cbc2f98f43bb03e61"},"motivation":"We conduct a systematic study of 18 widely used vision-language benchmarks and identify three major issues: 1) many items do not rely on visual cues and therefore fail to effectively measure multimodal understanding; 2) many items are already close to performance saturation for current LVLMs, which limits their discriminative power; 3) a small number of anomalous items affect the reliability of evaluation results.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22437","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mmhbench_02d53b0e","familyId":"bmf_756e91e3ea85","name":"MMHBench","oneLine":"MMHBench is a multimodal benchmark for mental health understanding in long-form videos, comprising 268 videos and 2,184 questions across third-person and first-person settings. It evaluates model reasoning about mental states from multimodal evidence.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27895","pdf":"https://arxiv.org/pdf/2607.27895","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.27895"},"evidence":{"snippet":"To address this limitation, we introduce MMHBench, a comprehensive multimodal benchmark for multi-perspective mental health understanding, comprising 268 long-form videos and 2,184 carefully curated questions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27895"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MMHBench is a multimodal benchmark for mental health understanding in long-form videos, comprising 268 videos and 2,184 questions across third-person and first-person settings. It evaluates model reasoning about mental states from multimodal evidence.","whyItMatters":"This benchmark targets a critical domain where existing evaluation is limited to coarse classification. It could provide finer-grained assessment of mental health understanding, but its current availability is unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9c8f40c7af40d058f987f2d913e6c9886c254aa59af9ff397c26f194785ee21b"},"motivation":"Mental health understanding in long-form videos requires nuanced reasoning over observable behavior, interpersonal context, and latent psychological states.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27895","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mmjailbench_1b3e3a1a","familyId":"bmf_0fe395ff23f1","name":"MMJailBench","oneLine":"Evaluates multimodal jailbreak vulnerabilities by factorizing harmful intent, prompt framing, visual semantics, and instruction carrier.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-26","firstSeenAt":"2026-08-27","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25490","pdf":"https://arxiv.org/pdf/2608.25490","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To address this limitation, we introduce MMJailBench, a factorized benchmark that systematically varies and combines these factors under controlled configurations, enabling fine-grained comparison and factor-level attribution.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25490"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates multimodal jailbreak vulnerabilities by factorizing harmful intent, prompt framing, visual semantics, and instruction carrier.","whyItMatters":"Enables factor-level attribution of MLLM safety failures, revealing dominant sources of vulnerability across harm domains.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"51d0ac08e9a23b0a64bb2fefbac9b8e81dfafc0bd437d89c2cec3b3b4b7ce112"},"motivation":"Multimodal Large Language Models (MLLMs) are increasingly deployed in real-world applications, yet how different factors shape their jailbreak vulnerabilities remains poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:12:36.511123Z","model":"deepseek-v4-pro","decisionReason":"MMJailBench is named and defined as a benchmark with a modular evaluation suite, multiple judge options, and multidimensional metrics; no explicit code link is provided.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce MMJailBench, a factorized benchmark that systematically varies and combines these factors under controlled configurations"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25490","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":75,"confidence":"Low","horizon":"7d","reason":"High-interest MLLM safety topic with 16 models evaluated, but lack of a public code or data URL may limit immediate reproducibility."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_3cc95f78bfacb570","familyId":"catalog_family_3cc95f78bfacb570","name":"MMLongBench-128K","oneLine":"MMLongBench-128K evaluates multimodal long-context understanding at a 128K token context length, testing how well vision-language models reason over very long mixed text and image inputs.","description":"MMLongBench-128K evaluates multimodal long-context understanding at a 128K token context length, testing how well vision-language models reason over very long mixed text and image inputs.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmlongbench-128k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3cc95f78bfacb570"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmlongbench-128k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmlongbench-128k","url":"https://llm-stats.com/benchmarks/mmlongbench-128k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","multimodal","reasoning","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_c3a53bd1982561d1","familyId":"catalog_family_c3a53bd1982561d1","name":"MMLongBench-Doc","oneLine":"MMLongBench-Doc evaluates long document understanding capabilities in vision-language models.","description":"MMLongBench-Doc evaluates long document understanding capabilities in vision-language models.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Long Context","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c3a53bd1982561d1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mmlongbenchdoc"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmlongbench-doc"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mmLongBenchDoc","url":"https://benchlm.ai/benchmarks/mmlongbenchdoc","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"MMLongBench-Doc","format":"Document-grounded reasoning","tasks":"Long document understanding","successorKey":null},{"catalog":"llm-stats","sourceId":"mmlongbench-doc","url":"https://llm-stats.com/benchmarks/mmlongbench-doc","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","long context","multimodal","vision"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_mmlongembed_b6d36e5b","familyId":"bmf_5fff974393cf","name":"MMLongEmbed","oneLine":"MMLongEmbed is a benchmark for evaluating multimodal embedding models in long-context scenarios, covering four retrieval tasks across text, document, and video modalities with varying context lengths.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.14747","pdf":"https://arxiv.org/pdf/2606.14747","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.14747"},"evidence":{"snippet":"To address the lack of systematic evaluation in this setting, we introduce MMLongEmbed, the first comprehensive benchmark for evaluating MEMs in long-context scenarios.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.14747"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MMLongEmbed is a benchmark for evaluating multimodal embedding models in long-context scenarios, covering four retrieval tasks across text, document, and video modalities with varying context lengths.","whyItMatters":"It addresses the gap in evaluating long-context multimodal embeddings, which is critical for real-world deployment where models must handle long inputs effectively.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5abd1a420682b613e5054f116592b3405bc6ad13c8d060fd8294d6932c113270"},"motivation":"Recent advancements have significantly expanded the theoretical context windows of Multimodal Embedding Models (MEMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.14747","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Long Context & Memory"],"domainScope":"general"},{"id":"lib_mmlu","familyId":"family_mmlu","name":"MMLU","oneLine":"Established benchmark family · Knowledge & Reasoning.","area":"Knowledge & Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge & Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2020-01-01","releaseDatePrecision":"year","firstRelease":{"year":2020,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2009.03300","pdf":null,"project":"https://github.com/hendrycks/test","code":"https://github.com/hendrycks/test","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_mmlu"},"ranking":{},"recordType":"family","aliases":["Massive Multitask Language Understanding"],"sourceAttribution":[{"role":"benchmark-paper","url":"https://arxiv.org/abs/2009.03300"}],"adoptionRefs":["openai-gpt5","anthropic-claude4","google-gemini25","deepseek-v3"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"},{"sourceId":"deepseek-v3","url":"https://github.com/deepseek-ai/DeepSeek-V3","provider":"DeepSeek","model":"DeepSeek-V3"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"mmlu","url":"https://benchlm.ai/benchmarks/mmlu","paperUrl":"https://arxiv.org/abs/2009.03300","year":"2020","fullName":"Massive Multitask Language Understanding","format":"Multiple choice questions","tasks":"57 subjects","successorKey":null},{"catalog":"llm-stats","sourceId":"mmlu","url":"https://llm-stats.com/benchmarks/mmlu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","language","legal","math","reasoning","finance","general","healthcare"],"catalogModelCount":101,"catalogStarCount":0},{"id":"catalog_f629c3f656123a31","familyId":"catalog_family_f629c3f656123a31","name":"MMLU (CoT)","oneLine":"Chain-of-Thought variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This version uses chain-of-thought prompting to elicit step-by-step reasoning.","description":"Chain-of-Thought variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This version uses chain-of-thought prompting to elicit step-by-step reasoning.","area":"Mathematical Reasoning","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Language","Legal","Math","Reasoning","Finance","General","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmlu-(cot)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f629c3f656123a31"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmlu-(cot)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmlu-(cot)","url":"https://llm-stats.com/benchmarks/mmlu-(cot)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","legal","math","reasoning","finance","general","healthcare"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"catalog_c35fcb235b7f78b2","familyId":"catalog_family_c35fcb235b7f78b2","name":"MMLU Chat","oneLine":"Chat-format variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This version uses conversational prompting format for model evaluation.","description":"Chat-format variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This version uses conversational prompting format for model evaluation.","area":"Mathematical Reasoning","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Language","Legal","Math","Reasoning","Finance","General","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmlu-chat","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c35fcb235b7f78b2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmlu-chat"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmlu-chat","url":"https://llm-stats.com/benchmarks/mmlu-chat","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","legal","math","reasoning","finance","general","healthcare"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"catalog_8c69d7c2b4104cfa","familyId":"catalog_family_8c69d7c2b4104cfa","name":"MMLU French","oneLine":"French language variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This multilingual version tests model performance in French.","description":"French language variant of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. This multilingual version tests model performance in French.","area":"Mathematical Reasoning","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Language","Legal","Math","Reasoning","Finance","General","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmlu-french","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8c69d7c2b4104cfa"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmlu-french"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmlu-french","url":"https://llm-stats.com/benchmarks/mmlu-french","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","legal","math","reasoning","finance","general","healthcare"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"catalog_6581924d125c4391","familyId":"catalog_family_6581924d125c4391","name":"MMLU-Base","oneLine":"Base version of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. Designed to comprehensively measure the breadth and depth of a model's academic and professional understanding.","description":"Base version of the Massive Multitask Language Understanding benchmark, evaluating language models across 57 tasks including elementary mathematics, US history, computer science, law, and other professional and academic subjects. Designed to comprehensively measure the breadth and depth of a model's academic and professional understanding.","area":"Mathematical Reasoning","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Language","Legal","Math","Reasoning","Finance","General","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmlu-base","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6581924d125c4391"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmlu-base"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmlu-base","url":"https://llm-stats.com/benchmarks/mmlu-base","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","legal","math","reasoning","finance","general","healthcare"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"lib_mmlu_pro","familyId":"family_mmlu","name":"MMLU-Pro","oneLine":"Established benchmark variant · Knowledge & Reasoning.","area":"Knowledge & Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge & Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2406.01574","pdf":null,"project":"https://github.com/TIGER-AI-Lab/MMLU-Pro","code":"https://github.com/TIGER-AI-Lab/MMLU-Pro","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_mmlu_pro"},"ranking":{},"recordType":"variant","aliases":[],"sourceAttribution":[{"role":"benchmark-paper","url":"https://arxiv.org/abs/2406.01574"}],"adoptionRefs":["openai-gpt5","google-gemini25"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"variantOf":"lib_mmlu","capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"mmluPro","url":"https://benchlm.ai/benchmarks/mmlu-pro","paperUrl":"https://arxiv.org/abs/2406.01574","year":"2024","fullName":"Massive Multitask Language Understanding Professional","format":"10-way multiple choice","tasks":"Multiple subjects","successorKey":null},{"catalog":"llm-stats","sourceId":"mmlu-pro","url":"https://llm-stats.com/benchmarks/mmlu-pro","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","language","legal","math","reasoning","finance","general","healthcare"],"catalogModelCount":135,"catalogStarCount":0},{"id":"catalog_8ea6e4d4b541df92","familyId":"catalog_family_8ea6e4d4b541df92","name":"MMLU-Pro (Arcee)","oneLine":"A display-only MMLU-Pro reference from Arcee AI's Trinity-Large-Thinking launch chart.","description":"A display-only MMLU-Pro reference from Arcee AI's Trinity-Large-Thinking launch chart.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.arcee.ai/blog/trinity-large-thinking","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8ea6e4d4b541df92"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mmluproarcee"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mmluProArcee","url":"https://benchlm.ai/benchmarks/mmluproarcee","paperUrl":"https://www.arcee.ai/blog/trinity-large-thinking","year":"2026","fullName":"MMLU-Pro first-party comparison snapshot","format":"10-way multiple choice","tasks":"Professional academic QA","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_5914fd3a09cc9c6b","familyId":"catalog_family_5914fd3a09cc9c6b","name":"MMLU-ProX","oneLine":"Extended version of MMLU-Pro providing additional challenging multiple-choice questions for evaluating language models across diverse academic and professional domains. Built on the foundation of the Massive Multitask Language Understanding benchmark framework.","description":"Extended version of MMLU-Pro providing additional challenging multiple-choice questions for evaluating language models across diverse academic and professional domains. Built on the foundation of the Massive Multitask Language Understanding benchmark framework.","area":"Mathematical Reasoning","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Multilingual","Language","Legal","Math","Reasoning","Finance","General","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2503.10497","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5914fd3a09cc9c6b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mmluprox"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmlu-prox"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mmluProX","url":"https://benchlm.ai/benchmarks/mmluprox","paperUrl":"https://arxiv.org/abs/2503.10497","year":"2025","fullName":"MMLU-ProX","format":"Multilingual multiple choice","tasks":"Multilingual professional QA","successorKey":null},{"catalog":"llm-stats","sourceId":"mmlu-prox","url":"https://llm-stats.com/benchmarks/mmlu-prox","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multilingual","language","legal","math","reasoning","finance","general","healthcare"],"catalogModelCount":32,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"catalog_ce91676a1a8503aa","familyId":"catalog_family_ce91676a1a8503aa","name":"MMLU-Redux","oneLine":"An improved version of the MMLU benchmark featuring manually re-annotated questions to identify and correct errors in the original dataset. Provides more reliable evaluation metrics for language models by addressing dataset quality issues found in the original MMLU.","description":"An improved version of the MMLU benchmark featuring manually re-annotated questions to identify and correct errors in the original dataset. Provides more reliable evaluation metrics for language models by addressing dataset quality issues found in the original MMLU.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Language","Math","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ce91676a1a8503aa"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mmluredux"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmlu-redux"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mmluRedux","url":"https://benchlm.ai/benchmarks/mmluredux","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"MMLU-Redux","format":"Multiple choice questions","tasks":"Broad academic QA","successorKey":null},{"catalog":"llm-stats","sourceId":"mmlu-redux","url":"https://llm-stats.com/benchmarks/mmlu-redux","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","language","math","reasoning","general"],"catalogModelCount":48,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_f6f392877d0ea787","familyId":"catalog_family_f6f392877d0ea787","name":"MMLU-redux-2.0","oneLine":"A curated version of the MMLU benchmark featuring manually re-annotated 5,700 questions across 57 subjects to identify and correct errors in the original dataset. Addresses the 6.49% error rate found in MMLU and provides more reliable evaluation metrics for language models.","description":"A curated version of the MMLU benchmark featuring manually re-annotated 5,700 questions across 57 subjects to identify and correct errors in the original dataset. Addresses the 6.49% error rate found in MMLU and provides more reliable evaluation metrics for language models.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Math","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmlu-redux-2.0","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f6f392877d0ea787"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmlu-redux-2.0"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmlu-redux-2.0","url":"https://llm-stats.com/benchmarks/mmlu-redux-2.0","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","math","reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_efa4bb0861df8733","familyId":"catalog_family_efa4bb0861df8733","name":"MMLU-STEM","oneLine":"STEM-focused subset of the Massive Multitask Language Understanding benchmark, evaluating language models on science, technology, engineering, and mathematics topics including physics, chemistry, mathematics, and other technical subjects.","description":"STEM-focused subset of the Massive Multitask Language Understanding benchmark, evaluating language models on science, technology, engineering, and mathematics topics including physics, chemistry, mathematics, and other technical subjects.","area":"Mathematical Reasoning","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Math","Physics","Reasoning","Chemistry"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmlu-stem","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_efa4bb0861df8733"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmlu-stem"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmlu-stem","url":"https://llm-stats.com/benchmarks/mmlu-stem","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","physics","reasoning","chemistry"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"catalog_39b617c84c45092e","familyId":"catalog_family_39b617c84c45092e","name":"MMMLU","oneLine":"Multilingual Massive Multitask Language Understanding dataset released by OpenAI, featuring professionally translated MMLU test questions across 14 languages including Arabic, Bengali, German, Spanish, French, Hindi, Indonesian, Italian, Japanese, Korean, Portuguese, Swahili, Yoruba, and Chinese. Contains approximately 15,908 multiple-choice questions per language covering 57 subjects.","description":"Multilingual Massive Multitask Language Understanding dataset released by OpenAI, featuring professionally translated MMLU test questions across 14 languages including Arabic, Bengali, German, Spanish, French, Hindi, Indonesian, Italian, Japanese, Korean, Portuguese, Swahili, Yoruba, and Chinese. Contains approximately 15,908 multiple-choice questions per language covering 57 subjects.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Language","Math","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/datasets/openai/MMMLU","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_39b617c84c45092e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mmmlu"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmmlu"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mmmlu","url":"https://benchlm.ai/benchmarks/mmmlu","paperUrl":"https://huggingface.co/datasets/openai/MMMLU","year":"2026","fullName":"MMMLU","format":"Exact match","tasks":"Multilingual academic QA","successorKey":null},{"catalog":"llm-stats","sourceId":"mmmlu","url":"https://llm-stats.com/benchmarks/mmmlu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","language","math","reasoning","general"],"catalogModelCount":49,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"lib_mmmu","familyId":"family_mmmu","name":"MMMU","oneLine":"Established benchmark family · Multimodal Understanding.","area":"Multimodal Understanding","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal Understanding"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2023-01-01","releaseDatePrecision":"year","firstRelease":{"year":2023,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2311.16502","pdf":null,"project":"https://mmmu-benchmark.github.io/","code":"https://github.com/MMMU-Benchmark/MMMU","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_mmmu"},"ranking":{},"recordType":"family","aliases":["Massive Multi-discipline Multimodal Understanding"],"sourceAttribution":[{"role":"official-project","url":"https://mmmu-benchmark.github.io/"}],"adoptionRefs":["openai-gpt5","anthropic-claude4","google-gemini25"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"mmmu","url":"https://benchlm.ai/benchmarks/mmmu","paperUrl":"https://arxiv.org/abs/2401.05508","year":"2024","fullName":"Massive Multi-discipline Multimodal Understanding","format":"Image + text question answering","tasks":"Multimodal academic reasoning","successorKey":null},{"catalog":"benchlm","sourceId":"valsMmmu","url":"https://benchlm.ai/benchmarks/valsmmmu","paperUrl":"https://www.vals.ai/benchmarks/mmmu","year":"2026","fullName":"Vals MMMU","format":"Accuracy score","tasks":"Multimodal academic task suite","successorKey":null},{"catalog":"llm-stats","sourceId":"mmmu","url":"https://llm-stats.com/benchmarks/mmmu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","external","multimodal","reasoning","general","healthcare","vision"],"catalogModelCount":64,"catalogStarCount":0},{"id":"catalog_2cc7090a8f862417","familyId":"catalog_family_2cc7090a8f862417","name":"MMMU (val)","oneLine":"Validation set for MMMU (Massive Multi-discipline Multimodal Understanding and Reasoning) benchmark, designed to evaluate multimodal models on massive multi-discipline tasks demanding college-level subject knowledge and deliberate reasoning across Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, and Tech & Engineering.","description":"Validation set for MMMU (Massive Multi-discipline Multimodal Understanding and Reasoning) benchmark, designed to evaluate multimodal models on massive multi-discipline tasks demanding college-level subject knowledge and deliberate reasoning across Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, and Tech & Engineering.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","General","Healthcare","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmmu-(val)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2cc7090a8f862417"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmmu-(val)"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmmuval"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmmu-(val)","url":"https://llm-stats.com/benchmarks/mmmu-(val)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false},{"catalog":"llm-stats","sourceId":"mmmuval","url":"https://llm-stats.com/benchmarks/mmmuval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","general","healthcare","vision"],"catalogModelCount":13,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_5ae4c639f19a8ff4","familyId":"catalog_family_5ae4c639f19a8ff4","name":"MMMU (validation)","oneLine":"Validation set of the Massive Multi-discipline Multimodal Understanding and Reasoning benchmark. Features college-level multimodal questions across 6 core disciplines (Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, Tech & Engineering) spanning 30 subjects and 183 subfields with diverse image types including charts, diagrams, maps, and tables.","description":"Validation set of the Massive Multi-discipline Multimodal Understanding and Reasoning benchmark. Features college-level multimodal questions across 6 core disciplines (Art & Design, Business, Science, Health & Medicine, Humanities & Social Science, Tech & Engineering) spanning 30 subjects and 183 subfields with diverse image types including charts, diagrams, maps, and tables.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","General","Healthcare","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmmu-(validation)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5ae4c639f19a8ff4"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmmu-(validation)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmmu-(validation)","url":"https://llm-stats.com/benchmarks/mmmu-(validation)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","general","healthcare","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"lib_mmmu_pro","familyId":"family_mmmu","name":"MMMU-Pro","oneLine":"Established benchmark variant · Multimodal Understanding.","area":"Multimodal Understanding","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal Understanding"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2409.02813","pdf":null,"project":"https://mmmu-benchmark.github.io/","code":"https://github.com/MMMU-Benchmark/MMMU","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_mmmu_pro"},"ranking":{},"recordType":"variant","aliases":["MMMU Pro"],"sourceAttribution":[{"role":"benchmark-paper","url":"https://arxiv.org/abs/2409.02813"}],"adoptionRefs":["openai-gpt5","google-gemini25"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"variantOf":"lib_mmmu","capabilityGroups":["Multimodal Perception"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"mmmuPro","url":"https://benchlm.ai/benchmarks/mmmu-pro","paperUrl":"https://arxiv.org/abs/2409.02813","year":"2024","fullName":"Massive Multi-discipline Multimodal Understanding Pro","format":"Image + text question answering","tasks":"Multimodal academic reasoning","successorKey":null},{"catalog":"llm-stats","sourceId":"mmmu-pro","url":"https://llm-stats.com/benchmarks/mmmu-pro","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","general","vision"],"catalogModelCount":69,"catalogStarCount":0},{"id":"catalog_edc9f28db2049b31","familyId":"catalog_family_edc9f28db2049b31","name":"MMMU-Pro (with tools)","oneLine":"MMMU-Pro variant evaluated with tool access enabled.","description":"MMMU-Pro variant evaluated with tool access enabled.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","General","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmmu-pro-with-tools","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_edc9f28db2049b31"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmmu-pro-with-tools"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmmu-pro-with-tools","url":"https://llm-stats.com/benchmarks/mmmu-pro-with-tools","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","general","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_e58dcf8ca58f20d6","familyId":"catalog_family_e58dcf8ca58f20d6","name":"MMMU-Pro w/ Python","oneLine":"Tool-augmented MMMU-Pro variant that allows Python assistance during multimodal reasoning.","description":"Tool-augmented MMMU-Pro variant that allows Python assistance during multimodal reasoning.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e58dcf8ca58f20d6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mmmupropython"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mmmuProPython","url":"https://benchlm.ai/benchmarks/mmmupropython","paperUrl":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","year":"2026","fullName":"MMMU-Pro with Python","format":"Image + text question answering with Python","tasks":"Multimodal academic reasoning","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_mmooc_7beeb312","familyId":"bmf_4c8571bb6649","name":"MMOOC","oneLine":"MMOOC evaluates multimodal large language models on out-of-context (OOC) and shifted in-context (Shifted IC) visual question answering. It contains over 41K image-question pairs covering three question formats, eight shift types, and six visual scenarios. Responses are scored for accuracy and refusal rate, with an LLM-as-a-judge metric for reasoning correctness.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27637","pdf":"https://arxiv.org/pdf/2607.27637","project":"https://zhuwenjie98.github.io/MMOOC-project-page/","code":"https://github.com/ZhuWenjie98/MMOOC","data":null,"hfPaper":"https://huggingface.co/papers/2607.27637"},"evidence":{"snippet":"To fill this gap, we present MMOOC, a large-scale benchmark for evaluating refusal and robust answering abilities of MLLMs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":6,"hfDailySubmittedAt":"2026-08-11T00:00:00.000Z","githubStars":31,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27637"},"ranking":{"90d":{"score":46,"rank":96,"coverage":0.7,"confidence":"Medium"}},"description":"MMOOC evaluates multimodal large language models on out-of-context (OOC) and shifted in-context (Shifted IC) visual question answering. It contains over 41K image-question pairs covering three question formats, eight shift types, and six visual scenarios. Responses are scored for accuracy and refusal rate, with an LLM-as-a-judge metric for reasoning correctness.","whyItMatters":"Existing benchmarks focus on unanswerable questions but overlook answerable shifted contexts. MMOOC provides a joint evaluation of refusal and robust answering, offering insight into model reliability in real-world scenarios where contexts are imperfect. This supports comparing models on a balanced measure of capability and safety.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"185953f78351ad4e6fd1aec2a3adc8e362b4ffa5bcc00ab38aa66c6f6989464e"},"motivation":"Multimodal Large Language Models (MLLMs) have achieved strong performance on a wide range of vision-language tasks, but often fail under imperfect or shifted contexts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27637","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MMOOC Project Team","organizationType":"academic-lab","sourceUrl":"https://zhuwenjie98.github.io/MMOOC-project-page/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mmpisa-bench_4787a595","familyId":"bmf_55650971ddf8","name":"mmPISA-bench","oneLine":"mmPISA-bench consists of 25 multiple-choice questions from PISA in 43 languages with official and machine translations, used to evaluate LLMs' reasoning across languages.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.07069","pdf":"https://arxiv.org/pdf/2606.07069","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07069"},"evidence":{"snippet":"We introduce mmPISA-bench, a compact high-quality multilingual reasoning benchmark derived from the OECD Programme for International Student Assessment (PISA).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07069"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"mmPISA-bench consists of 25 multiple-choice questions from PISA in 43 languages with official and machine translations, used to evaluate LLMs' reasoning across languages.","whyItMatters":"It addresses multilingual reasoning evaluation but the small scale and focus on proprietary models limit its utility as a general benchmark.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"288a487655e853ac5d9179878ebee29f6c6a920bc03cf1573e3696736b9dc25c"},"motivation":"We introduce mmPISA-bench, a compact high-quality multilingual reasoning benchmark derived from the OECD Programme for International Student Assessment (PISA).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07069","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_69c5c54539958b82","familyId":"catalog_family_69c5c54539958b82","name":"MMSearch","oneLine":"MMSearch evaluates multimodal models on search-based retrieval and question answering tasks that require processing both visual and textual information from search results.","description":"MMSearch evaluates multimodal models on search-based retrieval and question answering tasks that require processing both visual and textual information from search results.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Search","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_69c5c54539958b82"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mmsearch"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmsearch"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mmSearch","url":"https://benchlm.ai/benchmarks/mmsearch","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"MMSearch","format":"Mixed-media retrieval and grounded answering","tasks":"Multimodal search tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"mmsearch","url":"https://llm-stats.com/benchmarks/mmsearch","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","search","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_b3627d7b9e6c153e","familyId":"catalog_family_b3627d7b9e6c153e","name":"MMSearch-Plus","oneLine":"MMSearch-Plus is an extended variant of MMSearch with harder multimodal search and retrieval tasks requiring deeper reasoning over visual and textual search results.","description":"MMSearch-Plus is an extended variant of MMSearch with harder multimodal search and retrieval tasks requiring deeper reasoning over visual and textual search results.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Search","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b3627d7b9e6c153e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mmsearchplus"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmsearch-plus"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mmSearchPlus","url":"https://benchlm.ai/benchmarks/mmsearchplus","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"MMSearch-Plus","format":"Advanced mixed-media retrieval benchmark","tasks":"Hard multimodal search tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"mmsearch-plus","url":"https://llm-stats.com/benchmarks/mmsearch-plus","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","search","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Search & Retrieval"],"domainScope":"general"},{"id":"bm_mmshopbench_a4cdc83b","familyId":"bmf_b4f7e925e936","name":"MMShopBench","oneLine":"MMShopBench is a real-log benchmark for multimodal multi-turn shopping agents. It uses cleaned shopping logs with annotations for purchase intent and mandatory product requirements. Agents must infer requirements from images and dialogue, retrieve candidates, and verify product satisfaction.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.29002","pdf":"https://arxiv.org/pdf/2607.29002","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.29002"},"evidence":{"snippet":"We introduce MMShopBench, the first real-log benchmark for multimodal, multi-turn shopping agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.29002"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MMShopBench is a real-log benchmark for multimodal multi-turn shopping agents. It uses cleaned shopping logs with annotations for purchase intent and mandatory product requirements. Agents must infer requirements from images and dialogue, retrieve candidates, and verify product satisfaction.","whyItMatters":"Existing benchmarks often rely on text-only or synthetic requests, missing complex real-world multimodal shopping needs. MMShopBench provides a realistic evaluation to advance shopping agents, with an offline sandbox for reproducible experimentation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d1a96fc9628e62a6a2cedc0d6b03def4a0b9e8b0e9447f8634fabfecbce69001"},"motivation":"Online shoppers increasingly turn to AI shopping assistants, using images and multi-turn dialogue to express and refine product needs that are difficult to articulate in text alone.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.29002","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_8626ef0d34057617","familyId":"catalog_family_8626ef0d34057617","name":"MMSIBench","oneLine":"MMSIBench is a multimodal spatial-intelligence benchmark evaluating spatial reasoning over images.","description":"MMSIBench is a multimodal spatial-intelligence benchmark evaluating spatial reasoning over images.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Spatial Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmsibench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8626ef0d34057617"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmsibench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmsibench","url":"https://llm-stats.com/benchmarks/mmsibench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","spatial reasoning","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_377aca97e89d030c","familyId":"catalog_family_377aca97e89d030c","name":"MMStar","oneLine":"MMStar is an elite vision-indispensable multimodal benchmark comprising 1,500 challenge samples meticulously selected by humans to evaluate 6 core capabilities and 18 detailed axes. The benchmark addresses issues of visual content unnecessity and unintentional data leakage in existing multimodal evaluations.","description":"MMStar is an elite vision-indispensable multimodal benchmark comprising 1,500 challenge samples meticulously selected by humans to evaluate 6 core capabilities and 18 detailed axes. The benchmark addresses issues of visual content unnecessity and unintentional data leakage in existing multimodal evaluations.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","General","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmstar","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_377aca97e89d030c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmstar"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmstar","url":"https://llm-stats.com/benchmarks/mmstar","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","general","vision"],"catalogModelCount":25,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_3b039d0c4f726cef","familyId":"catalog_family_3b039d0c4f726cef","name":"MMT-Bench","oneLine":"MMT-Bench is a comprehensive multimodal benchmark for evaluating Large Vision-Language Models towards multitask AGI. It comprises 31,325 meticulously curated multi-choice visual questions from various multimodal scenarios such as vehicle driving and embodied navigation, covering 32 core meta-tasks and 162 subtasks in multimodal understanding.","description":"MMT-Bench is a comprehensive multimodal benchmark for evaluating Large Vision-Language Models towards multitask AGI. It comprises 31,325 meticulously curated multi-choice visual questions from various multimodal scenarios such as vehicle driving and embodied navigation, covering 32 core meta-tasks and 162 subtasks in multimodal understanding.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","General","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmt-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3b039d0c4f726cef"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmt-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmt-bench","url":"https://llm-stats.com/benchmarks/mmt-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","general","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_d580a014be7a9964","familyId":"catalog_family_d580a014be7a9964","name":"MMVet","oneLine":"MM-Vet is an evaluation benchmark that examines large multimodal models on complicated multimodal tasks requiring integrated capabilities. It assesses six core vision-language capabilities: recognition, knowledge, spatial awareness, language generation, OCR, and math through questions that require one or more of these capabilities.","description":"MM-Vet is an evaluation benchmark that examines large multimodal models on complicated multimodal tasks requiring integrated capabilities. It assesses six core vision-language capabilities: recognition, knowledge, spatial awareness, language generation, OCR, and math through questions that require one or more of these capabilities.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Multimodal","Reasoning","Spatial Reasoning","General","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmvet","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d580a014be7a9964"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmvet"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmvet","url":"https://llm-stats.com/benchmarks/mmvet","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","multimodal","reasoning","spatial reasoning","general","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_93d1c075bf34e12c","familyId":"catalog_family_93d1c075bf34e12c","name":"MMVetGPT4Turbo","oneLine":"MM-Vet evaluation using GPT-4 Turbo for scoring. This variant of MM-Vet examines large multimodal models on complicated multimodal tasks requiring integrated capabilities across six core vision-language abilities: recognition, knowledge, spatial awareness, language generation, OCR, and math.","description":"MM-Vet evaluation using GPT-4 Turbo for scoring. This variant of MM-Vet examines large multimodal models on complicated multimodal tasks requiring integrated capabilities across six core vision-language abilities: recognition, knowledge, spatial awareness, language generation, OCR, and math.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Multimodal","Reasoning","Spatial Reasoning","General","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mmvetgpt4turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_93d1c075bf34e12c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmvetgpt4turbo"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mmvetgpt4turbo","url":"https://llm-stats.com/benchmarks/mmvetgpt4turbo","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","multimodal","reasoning","spatial reasoning","general","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_1a552b9a0512cd49","familyId":"catalog_family_1a552b9a0512cd49","name":"MMVU","oneLine":"MMVU (Multimodal Multi-disciplinary Video Understanding) is a benchmark for evaluating multimodal models on video understanding tasks across multiple disciplines, testing comprehension and reasoning capabilities on video content.","description":"MMVU (Multimodal Multi-disciplinary Video Understanding) is a benchmark for evaluating multimodal models on video understanding tasks across multiple disciplines, testing comprehension and reasoning capabilities on video content.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Reasoning","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k2-5.html","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1a552b9a0512cd49"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mmvu"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mmvu"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mmvu","url":"https://benchlm.ai/benchmarks/mmvu","paperUrl":"https://www.kimi.com/blog/kimi-k2-5.html","year":"2026","fullName":"Multimodal Multi-disciplinary Video Understanding","format":"Video reasoning benchmark","tasks":"Video understanding","successorKey":null},{"catalog":"llm-stats","sourceId":"mmvu","url":"https://llm-stats.com/benchmarks/mmvu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","video","vision"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_can-language-models-understand-mmwave-data_7db61a9b","familyId":"bmf_af5a9cee9c32","name":"mmWave-QA","oneLine":"mmWave-QA is a benchmark for language-conditioned mmWave human perception, aggregating heterogeneous public datasets into six scenarios and five QA tasks for standardized evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.14179","pdf":"https://arxiv.org/pdf/2608.14179","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Building on this, we present mmWave-QA, the first benchmark for language-conditioned mmWave human perception.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14179"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"mmWave-QA is a benchmark for language-conditioned mmWave human perception, aggregating heterogeneous public datasets into six scenarios and five QA tasks for standardized evaluation.","whyItMatters":"It enables standardized evaluation across diverse mmWave hardware and conditions, filling a gap in radar-language integration.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"3b69b023b1878a620c1ad7e98e84e562fc84e15ca753f0e382b5cebbfc3c4837"},"motivation":"Large language models (LLMs) have shown remarkable reasoning and generative capabilities, motivating their use as universal reasoning engines for perception.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The abstract explicitly names the benchmark and describes a stable public evaluation protocol.","canonicalNameSource":"abstract","canonicalNameEvidence":"we present mmWave-QA, the first benchmark for language-conditioned mmWave human perception"},"publication":{"status":"publication_reported","venue":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Findings, 2026","evidence":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Findings, 2026","evidenceUrl":"https://arxiv.org/abs/2608.14179","source":"arxiv-journal-reference","evidenceLevel":"strong-author-metadata","verifiedAt":"2026-08-19T11:05:13.395391Z"},"publications":[{"venueName":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Findings, 2026","publicationStatus":"published","evidence":[{"sourceType":"arxiv-journal-reference","sourceUrl":"https://arxiv.org/abs/2608.14179","observedAt":"2026-08-19T11:05:13.395391Z","rawValue":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Findings, 2026","level":"strong-author-metadata"}]}],"attentionForecast":{"score":50,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets a niche but emerging mmWave-LLM area with a named dataset, suggesting moderate attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_mobilejudgebench_5adbb014","familyId":"bmf_482470db024a","name":"MobileJudgeBench","oneLine":"MobileJudgeBench evaluates LLM-as-judge methods on mobile agent trajectories. It includes 931 human-annotated trajectories from 6 mobile agent benchmarks, covering 4 agent models and 68 apps.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.11434","pdf":"https://arxiv.org/pdf/2608.11434","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.11434"},"evidence":{"snippet":"We introduce MobileJudgeBench, a benchmark for systematically evaluating LLM-as-judge methods on mobile agent trajectories.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11434"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MobileJudgeBench evaluates LLM-as-judge methods on mobile agent trajectories. It includes 931 human-annotated trajectories from 6 mobile agent benchmarks, covering 4 agent models and 68 apps.","whyItMatters":"Mobile agent benchmarks increasingly rely on LLM-based judges, yet their reliability is unexamined. MobileJudgeBench fills this gap by providing a standardized evaluation to select reliable judges, improving evaluation fidelity and downstream reinforcement learning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5d44e8805339c52a96448ca033dacbe21acbd61156a6ee19589c1e558af7fb54"},"motivation":"Mobile agent benchmarks increasingly rely on LLM-based judges to evaluate task completion, yet the reliability of these judges on mobile agent trajectories remains largely unexamined.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11434","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_846e7f8f5af22dad","familyId":"catalog_family_846e7f8f5af22dad","name":"MobileMiniWob++_SR","oneLine":"MobileMiniWob++ SR (Success Rate) is an adaptation of the MiniWob++ web interaction benchmark for mobile Android environments within AndroidWorld. It comprises 92 web interaction tasks adapted for touch-based mobile interfaces, evaluating agents' ability to navigate and interact with web applications on mobile devices.","description":"MobileMiniWob++ SR (Success Rate) is an adaptation of the MiniWob++ web interaction benchmark for mobile Android environments within AndroidWorld. It comprises 92 web interaction tasks adapted for touch-based mobile interfaces, evaluating agents' ability to navigate and interact with web applications on mobile devices.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Frontend Development","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mobileminiwob++-sr","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_846e7f8f5af22dad"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mobileminiwob++-sr"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mobileminiwob++-sr","url":"https://llm-stats.com/benchmarks/mobileminiwob++-sr","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","frontend development","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_mobilepa-bench_6f002001","familyId":"bmf_f55e72fcab83","name":"MobilePA-Bench","oneLine":"Evaluates mobile planning agents on tool-calling and planning across 13 domains and 212 tools in an interactive sandbox, with scoring on three advanced planning dimensions.","area":"Agents & Tool Use","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Planning"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.23035","pdf":"https://arxiv.org/pdf/2608.23035","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To close this gap, we present \\textbf{MobilePA-Bench}, an interactive, stateful, and tool-centric benchmark for evaluating the tool-calling and planning abilities of mobile planning agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":41,"hfDailySubmittedAt":"2026-08-25T00:00:00.000Z","githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23035"},"ranking":{"30d":{"score":39,"rank":52,"coverage":0.85,"confidence":"High"},"90d":{"score":35,"rank":194,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates mobile planning agents on tool-calling and planning across 13 domains and 212 tools in an interactive sandbox, with scoring on three advanced planning dimensions.","whyItMatters":"Closes the gap between GUI-centric and static benchmarks by providing a runtime-grounded evaluation of mobile agents, highlighting reliability issues in frontier LLMs.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"86323252b768a092b476642be876d44597abc4734892842918900ab7ecd52ce3"},"motivation":"As on-device LLM agents evolve into personal copilots, the mobile operating system has become a key testbed for this paradigm, making rigorous capability evaluation essential.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally introduced with a clear evaluation framework and artifact evidence, qualifying as an ongoing public benchmark.","canonicalNameSource":"abstract","canonicalNameEvidence":"we present \\textbf{MobilePA-Bench}, an interactive, stateful, and tool-centric benchmark for evaluating the tool-calling and planning abilities of mobile planning agents."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.23035","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets the popular mobile agent domain with a comprehensive, executable setup, likely to draw attention from researchers and practitioners in LLM agents."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"catalog_d344e4a4e7791f4f","familyId":"catalog_family_d344e4a4e7791f4f","name":"MobileWorld","oneLine":"MobileWorld is a benchmark for evaluating multimodal agents on real mobile-device tasks, testing GUI grounding, navigation, and multi-step task completion in mobile environments.","description":"MobileWorld is a benchmark for evaluating multimodal agents on real mobile-device tasks, testing GUI grounding, navigation, and multi-step task completion in mobile environments.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Multimodal","Agents","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.8","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d344e4a4e7791f4f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mobileworld"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mobileworld"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mobileWorld","url":"https://benchlm.ai/benchmarks/mobileworld","paperUrl":"https://qwen.ai/blog?id=qwen3.8","year":"2026","fullName":"MobileWorld","format":"Mobile agent task score","tasks":"Interactive mobile-device workflows","successorKey":null},{"catalog":"llm-stats","sourceId":"mobileworld","url":"https://llm-stats.com/benchmarks/mobileworld","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","multimodal","agents","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_mobileworldsafety_b2abd1b3","familyId":"bmf_d8688a41bdaf","name":"MobileWorldSafety","oneLine":"Evaluates Android GUI-agent safety against environmental injection attacks embedded in everyday mobile workflows.","area":"Agents & Tool Use","applicationDomains":["Consumer & Productivity","Cybersecurity"],"primaryDomain":"Consumer & Productivity","industrySectors":["Consumer Technology"],"capabilities":["Computer use","Prompt-injection resistance","Mobile interaction"],"topics":["Agents","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.17659","pdf":"https://arxiv.org/pdf/2608.17659","project":"https://anonymous.4open.science/r/Anonymous_sub-C887","code":"https://anonymous.4open.science/r/Anonymous_sub-C887","data":"https://anonymous.4open.science/r/Anonymous_sub-C887","hfPaper":"https://huggingface.co/papers/2608.17659"},"evidence":{"snippet":"To address this gap, we introduce MobileWorldSafety, a benchmark of 142 risk tasks built on real Android applications.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17659"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MobileWorldSafety evaluates GUI agents' safety against environmental injection attacks in Android apps. It includes 142 risk tasks on real applications, with programmatically verifiable risk indicators and a two-stage pipeline for verification.","whyItMatters":"Provides a quantitative measure of GUI agent vulnerability to injection attacks, enabling comparison across agents and supporting development of safer mobile agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2db901229ee0346033001fb1dcb72bf015d9f310f65a1d28077cd4fc2e1e1d70"},"motivation":"LLM-powered GUI agents that autonomously operate smartphones are rapidly transitioning from research prototypes to early real-world deployment.","constructionDetail":"MobileWorldSafety embeds environmental injection attacks into executable Android workflows and verifies both attack success and task completion.","detail":{"taskBreakdown":["Email","Messaging","Calendar","Files","Web browsing","Navigation","Collaboration","Social media","E-commerce","Tool responses"],"protocol":{"tasks":"142 risk tasks across 13 Android apps and 5 MCP servers","primaryMetric":"Attack Success Rate and Task Completion Rate under attack","version":"arXiv v1"}},"curation":{"state":"source-reviewed","reviewedAt":"2026-08-20","sources":["https://arxiv.org/abs/2608.17659","https://arxiv.org/html/2608.17659","https://anonymous.4open.science/r/Anonymous_sub-C887"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17659","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"score_submission","availability":{"paperStatus":"available","githubStatus":"available","hfDatasetStatus":"available","evaluatorStatus":"available","submissionStatus":"not_found"},"capabilityGroups":["Agents"],"domainScope":"cross-domain"},{"id":"bm_mohallbench_4062fe02","familyId":"bmf_4bf5148e0196","name":"MoHallBench","oneLine":"MoHallBench evaluates motion hallucination in video LLMs with 11,306 video clips and 40,493 QA pairs, covering three hallucination sources and multiple choice settings.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.01117","pdf":"https://arxiv.org/pdf/2607.01117","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.01117"},"evidence":{"snippet":"We present MoHallBench, a benchmark for diagnosing motion hallucination in VideoLLMs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01117"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MoHallBench evaluates motion hallucination in video LLMs with 11,306 video clips and 40,493 QA pairs, covering three hallucination sources and multiple choice settings.","whyItMatters":"Targets a specific video understanding failure mode, offering metrics to reduce affirmation bias and revealing gaps in action recognition vs. hallucination resistance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6e809959fb7db64d9322a2f884fdf69d38628c2724cc2878fc814cd47cf48147"},"motivation":"Video Large Language Models (VideoLLMs) have shown strong progress in video understanding, yet they still suffer from hallucinations that are inconsistent with visual evidence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01117","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_molsafeeval_ff6a51a3","familyId":"bmf_d128333456b3","name":"MolSafeEval","oneLine":"MolSafeEval evaluates safety risks in AI-generated molecules across four task types, using structured knowledge graphs and LLM-based reasoning to detect unsafe features.","area":"Safety & Trustworthiness","applicationDomains":["Science & Research","Health & Life Sciences"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals","Pharma & Biotech"],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.00464","pdf":"https://arxiv.org/pdf/2607.00464","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.00464"},"evidence":{"snippet":"To address this gap, we introduce MolSafeEval, a benchmark dedicated to evaluating and analyzing the safety risks of molecular generation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.00464"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MolSafeEval evaluates safety risks in AI-generated molecules across four task types, using structured knowledge graphs and LLM-based reasoning to detect unsafe features.","whyItMatters":"Fills a gap in molecular generation benchmarks by focusing on safety risks, providing standardized datasets and protocols for safer molecular design.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"110ba94d96f9b54c7265804078c5d2318e72423dcd38b2914ca6ecb9b3ef59ea"},"motivation":"Current molecular generation benchmarks emphasize task complexity, molecule novelty, and property alignment; they largely overlook a critical concern: the potential safety risks of AI-generated molecules.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"Findings of ACL 2026","evidence":"Accepted by Findings of ACL 2026","evidenceUrl":"https://arxiv.org/abs/2607.00464","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"Findings of ACL 2026","reviewStatus":"accepted","decisionRaw":"Accepted by Findings of ACL 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.00464","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by Findings of ACL 2026","level":"author-claim"}]}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"cross-domain"},{"id":"bm_moment-video_26d94dc0","familyId":"bmf_3b12aecab48e","name":"Moment-Video","oneLine":"Moment-Video evaluates video MLLMs on momentary visual event understanding through 1,000 human-verified video-QA pairs across four task types: temporal occurrence, temporal counting, action description, and temporal reasoning.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02522","pdf":"https://arxiv.org/pdf/2606.02522","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02522"},"evidence":{"snippet":"We introduce Moment-Video, a benchmark for diagnosing the temporal fidelity of video MLLMs through momentary visual event understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":13,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02522"},"ranking":{},"description":"Moment-Video evaluates video MLLMs on momentary visual event understanding through 1,000 human-verified video-QA pairs across four task types: temporal occurrence, temporal counting, action description, and temporal reasoning.","whyItMatters":"It addresses the gap in evaluating models' ability to capture brief answer-critical visual evidence, which is common in practical video understanding and not covered by general video benchmarks. The results show significant room for improvement, offering a diagnostic tool for model development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"de72bb5d2df58b3152d36085960cec92ec1d9654f413e113c8033b53390744aa"},"motivation":"Video multimodal large language models (MLLMs) have made rapid progress on general and long-form video understanding, yet their ability to preserve brief answer-critical visual evidence remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02522","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_momento_3a4ed98e","familyId":"bmf_70f4f3d5e856","name":"Momento","oneLine":"Momento benchmarks persistent agentic task completion in multi-session service environments, requiring agents to resolve temporal dependencies and evolving user goals across sessions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00832","pdf":"https://arxiv.org/pdf/2606.00832","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.00832"},"evidence":{"snippet":"We introduce Momento, a benchmark for persistent agentic task completion in multi-session service environments, requiring agents to take consequential, tool-mediated actions while resolving temporal dependencies and evolving user goals across sessions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00832"},"ranking":{},"description":"Momento benchmarks persistent agentic task completion in multi-session service environments, requiring agents to resolve temporal dependencies and evolving user goals across sessions.","whyItMatters":"Highlights the gap in agent evaluation by focusing on multi-session history and misestimation of user state, which is critical for realistic human-agent interaction.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6674810005bcb648bb98c33f2b9c2f56abd6c150c1a6a3de27e1dab3c76085ce"},"motivation":"Recent advances in agentic AI have enabled agents to complete complex tasks through tool use, reasoning, and multi-step planning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00832","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_more_09a372c7","familyId":"bmf_187897ce0afc","name":"MORE","oneLine":"MORE evaluates multilingual document parsing across 149 languages, covering text, formulas, tables, code blocks, catalogs, and reading order, using 1,288 real-world document images with human-refined annotations.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.02956","pdf":"https://arxiv.org/pdf/2607.02956","project":null,"code":"https://github.com/zimoqingfeng/MORE","data":null,"hfPaper":"https://huggingface.co/papers/2607.02956"},"evidence":{"snippet":"To bridge this gap, we introduce MORE, a large-scale benchmark designed for multilingual document parsing evaluation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02956"},"ranking":{"90d":{"score":30,"rank":246,"coverage":0.7,"confidence":"Medium"}},"description":"MORE evaluates multilingual document parsing across 149 languages, covering text, formulas, tables, code blocks, catalogs, and reading order, using 1,288 real-world document images with human-refined annotations.","whyItMatters":"Existing benchmarks focus on high-resource languages, leaving a gap in assessing multilingual document parsers. MORE provides a broad and structured evaluation to compare model capabilities on low-resource languages and diverse document elements.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"de191e91e618d2ef15e6d3be694619cb5aac0ea95ba4b04d80f106157c7c42e5"},"motivation":"Multilingual documents encapsulate rich regional cultures, scientific discoveries, and historical records.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"43rd International Conference on Machine Learning (ICML 2026)","evidence":"Accepted to the 43rd International Conference on Machine Learning (ICML 2026). 22 pages, 11 figures. Code and dataset available at https://github.com/zimoqingfeng/MORE","evidenceUrl":"https://arxiv.org/abs/2607.02956","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"43rd International Conference on Machine Learning (ICML 2026)","reviewStatus":"accepted","decisionRaw":"Accepted to the 43rd International Conference on Machine Learning (ICML 2026). 22 pages, 11 figures. Code and dataset available at https://github.com/zimoqingfeng/MORE","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.02956","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to the 43rd International Conference on Machine Learning (ICML 2026). 22 pages, 11 figures. Code and dataset available at https://github.com/zimoqingfeng/MORE","level":"author-claim"}]}],"publishers":[{"name":"MORE Benchmark Team","organizationType":"community","sourceUrl":"https://github.com/zimoqingfeng/MORE","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mortarbench_63bbc02d","familyId":"bmf_765485d41ac0","name":"MortarBench","oneLine":"MortarBench evaluates LLM-based agents on mortgage loan origination tasks, covering application, underwriting, approval, and funding. It uses synthetic data with edge-case coverage and exact match accuracy scoring.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.19416","pdf":"https://arxiv.org/pdf/2606.19416","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.19416"},"evidence":{"snippet":"To fill this gap, we present MortarBench, a loan origination agent benchmark.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19416"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MortarBench evaluates LLM-based agents on mortgage loan origination tasks, covering application, underwriting, approval, and funding. It uses synthetic data with edge-case coverage and exact match accuracy scoring.","whyItMatters":"Mortgage loan origination is a critical financial process increasingly augmented by LLMs, yet no public benchmark existed. MortarBench provides a standardized evaluation to measure agent performance and identify biases in this domain.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fa0989c2db3bf72af010a4cd81d6225c9248d2df6b753754a24d8b2e3fbe61a0"},"motivation":"Loan origination is the process by which a lender creates a new loan, from application and underwriting through approval and funding.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19416","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_c54b44e78bbf5ace","familyId":"catalog_family_c54b44e78bbf5ace","name":"MortgageTax","oneLine":"Vals AI benchmark for mortgage and tax document reasoning, including semantic and numerical extraction task views.","description":"Vals AI benchmark for mortgage and tax document reasoning, including semantic and numerical extraction task views.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/mortgage_tax","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c54b44e78bbf5ace"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsmortgagetax"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsMortgageTax","url":"https://benchlm.ai/benchmarks/valsmortgagetax","paperUrl":"https://www.vals.ai/benchmarks/mortgage_tax","year":"2026","fullName":"Vals MortgageTax","format":"Accuracy score","tasks":"Mortgage and tax extraction tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_beyond-global-scalars-synergizing-token-le_0d7c6f3e","familyId":"bmf_8649e8bb3d3b","name":"MOSAIC","oneLine":"Adversarial AIGC text detection benchmark with 16,000 samples spanning a full-granularity attack spectrum.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.28009","pdf":"https://arxiv.org/pdf/2608.28009","project":null,"code":"https://github.com/TencentBAC/NeuroStat","data":null,"hfPaper":null},"evidence":{"snippet":"To expose these flaws, we introduce MOSAIC, a comprehensive adversarial benchmark comprising 16000 samples across a full-granularity attack spectrum.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.28009"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Adversarial AIGC text detection benchmark with 16,000 samples spanning a full-granularity attack spectrum.","whyItMatters":"Addresses vulnerabilities of global statistical and semantic-only detectors by providing token-level adversarial scenarios for robust machine-generated text detection.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T04:51:30.440525Z","inputHash":"2f8efa37cead722fd96bab838c25ac02dc7dc6c8bfb83e256bbcbc6a8ecd9ff3"},"motivation":"The rapid evolution of large language models necessitates robust machine-generated text detection.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T04:51:30.440525Z","model":"deepseek-v4-pro","decisionReason":"Formally named MOSAIC in the abstract, with explicit code and benchmark release link, defining repeatable evaluation and scoring.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce MOSAIC, a comprehensive adversarial benchmark comprising 16000 samples across a full-granularity attack spectrum."},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 Findings","evidence":"Accepted by EMNLP 2026 Findings","evidenceUrl":"https://arxiv.org/abs/2608.28009","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T01:03:30.163531Z"},"venueAttempts":[{"venueName":"EMNLP 2026 Findings","reviewStatus":"accepted","decisionRaw":"Accepted by EMNLP 2026 Findings","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.28009","observedAt":"2026-08-31T01:03:30.163531Z","rawValue":"Accepted by EMNLP 2026 Findings","level":"author-claim"}]}],"attentionForecast":{"score":45,"confidence":"Medium","horizon":"7d","reason":"Accepted at EMNLP Findings and targeting adversarial text detection, a topic with broad community interest, but niche focus may limit initial reach."},"evaluationMode":"public_reusable","publishers":[{"name":"Tencent BAC","organizationType":"company-research-lab","sourceUrl":"https://github.com/TencentBAC/NeuroStat","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mosaicleaks_86df075c","familyId":"bmf_4525a035d410","name":"MosaicLeaks","oneLine":"MosaicLeaks is a benchmark of 1,001 multi-hop research tasks to evaluate privacy leakage in deep research agents, but no artifacts are provided in this article.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30727","pdf":"https://arxiv.org/pdf/2605.30727","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30727"},"evidence":{"snippet":"We introduce MosaicLeaks, a benchmark of 1,001 multi-hop deep research tasks that chain private enterprise documents and a public web corpus, forcing agents to make external queries that depend on local information.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30727"},"ranking":{},"description":"MosaicLeaks is a benchmark of 1,001 multi-hop research tasks to evaluate privacy leakage in deep research agents, but no artifacts are provided in this article.","whyItMatters":"It addresses privacy risks in agentic workflows, but without a public release, external teams cannot use it.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d37259aa71e04df97907b0a570aa239b2c522b1881c6fd1a62289435fd64fca3"},"motivation":"Deep research agents increasingly combine private local documents with external tools like web retrieval, creating a privacy risk: an agent's external queries may leak sensitive information from its local context.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30727","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_92f572dbd59f7255","familyId":"catalog_family_92f572dbd59f7255","name":"MotionBench","oneLine":"MotionBench is a benchmark for evaluating multimodal models on motion understanding in videos, testing the ability to comprehend temporal dynamics, movement patterns, and action sequences.","description":"MotionBench is a benchmark for evaluating multimodal models on motion understanding in videos, testing the ability to comprehend temporal dynamics, movement patterns, and action sequences.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/motionbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_92f572dbd59f7255"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/motionbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"motionbench","url":"https://llm-stats.com/benchmarks/motionbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","video","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_motionhalluc_82d851ff","familyId":"bmf_5aa2d058a5cc","name":"MotionHalluc","oneLine":"MotionHalluc is a benchmark evaluating kinematic hallucinations in cross-video motion comparison. It includes 1540 questions over 553 video pairs, assessing directional, attributional, and temporal hallucinations in generated instructions.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.23061","pdf":"https://arxiv.org/pdf/2606.23061","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.23061"},"evidence":{"snippet":"To systematically investigate these hallucinations, we introduce MotionHalluc, a dedicated benchmark for evaluating motion hallucinations in paired-video comparison.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.23061"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MotionHalluc is a benchmark evaluating kinematic hallucinations in cross-video motion comparison. It includes 1540 questions over 553 video pairs, assessing directional, attributional, and temporal hallucinations in generated instructions.","whyItMatters":"Large multimodal models often produce motion hallucinations in paired-video comparison tasks. A systematic benchmark like MotionHalluc could help measure and reduce these errors, potentially improving the reliability of automated motion feedback systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d3f82b13758784501626b4e9040b92eb94966af01ac8e72cf835d080fa5fe9bb"},"motivation":"Motion instruction generation in cross-video comparison aims to produce corrective feedback that describes the differences between a query and a reference motion.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.23061","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mov-bench_db8a731e","familyId":"bmf_5a9bcf5cba7a","name":"MOV-Bench","oneLine":"MOV-Bench contains 519 questions requiring multi-hop reasoning over temporally dispersed audio-visual evidence, evaluating omni-modal LLMs on cross-modal reasoning tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.28192","pdf":"https://arxiv.org/pdf/2605.28192","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28192"},"evidence":{"snippet":"In this work, we introduce MOV-Bench, a benchmark containing 519 carefully curated questions that require multi-hop reasoning over temporally dispersed audio-visual evidence.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28192"},"ranking":{},"description":"MOV-Bench contains 519 questions requiring multi-hop reasoning over temporally dispersed audio-visual evidence, evaluating omni-modal LLMs on cross-modal reasoning tasks.","whyItMatters":"Existing benchmarks offer limited investigation of multi-hop audio-visual reasoning; MOV-Bench provides a focused evaluation set for this capability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c097da2aa1043f63dd1e2aeae4f8aab528bb6dc2480d3cb2ba68281e33fb4f62"},"motivation":"Multi-hop audio-visual reasoning remains challenging for Omni-LLMs, as relevant evidence is often sparse, temporally dispersed, and distributed across both audio and visual streams.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28192","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_moving-alphabet_5bd200ea","familyId":"bmf_9c1339291416","name":"Moving Alphabet","oneLine":"Moving Alphabet is a procedural testbed for controlled experiments on text-to-video training data, generating synthetic videos with ground-truth metadata to study data distribution and caption quality effects.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.18789","pdf":"https://arxiv.org/pdf/2607.18789","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.18789"},"evidence":{"snippet":"To enable controlled experiments, we introduce Moving Alphabet, a procedural testbed that renders letters with varying fonts, colors, sizes, and positions, moving in different directions and speeds against a black background.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.18789"},"ranking":{"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Moving Alphabet is a procedural testbed for controlled experiments on text-to-video training data, generating synthetic videos with ground-truth metadata to study data distribution and caption quality effects.","whyItMatters":"It provides insights into data curation for text-to-video models, highlighting the importance of diverse distributions and caption quality, but it is not a model comparison benchmark.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2354ee3f9f8ea9b6123774383feef61664bc9ab43e70622b142ddf0dbd5c59c8"},"motivation":"Text-to-video generation has advanced significantly over the past five years through scaling of model size, data, and compute.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.18789","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mpar-bench_22a7bcc7","familyId":"bmf_346c45b511f3","name":"MPAR-Bench","oneLine":"MPAR-Bench evaluates multi-point associative reasoning in LLMs across English and Chinese using 1,000 items with diverse clues. Scoring includes exact-match accuracy, ANLS, embedding similarity, and reasoning-trace verification, with four perturbation types.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.10444","pdf":"https://arxiv.org/pdf/2608.10444","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.10444"},"evidence":{"snippet":"We introduce MPAR-Bench, a bilingual English-Chinese benchmark that isolates reasoning breadth through multi-point associative reasoning.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10444"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MPAR-Bench evaluates multi-point associative reasoning in LLMs across English and Chinese using 1,000 items with diverse clues. Scoring includes exact-match accuracy, ANLS, embedding similarity, and reasoning-trace verification, with four perturbation types.","whyItMatters":"Current benchmarks focus on reasoning depth (longer chains) but neglect breadth (parallel semantic exploration). MPAR-Bench fills that gap, showing that depth does not guarantee robust breadth, and offers a practical tool for assessing models on a complementary reasoning dimension.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2a8fdf05a2930e2a1a4ae4ff5f395ca2069bd53026909e9c4adc60208cdfc2e2"},"motivation":"Large language models (LLMs) have made substantial progress on reasoning tasks that require increasingly long and complex inferential chains.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10444","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"No official publisher identified","organizationType":"community","sourceUrl":"https://arxiv.org/abs/2608.10444","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_mpc-patch-bench_574ee054","familyId":"bmf_0c668e75b30f","name":"MPC-Patch-Bench","oneLine":"MPC-Patch-Bench evaluates LLM-based code repair on repository-level Secure Multi-Party Computation (MPC) software. It provides 205 verified instances with Fail-to-Pass/Pass-to-Pass tests and a verifier that checks cryptographic safety and numerical fidelity via differential testing and static analysis.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.CR"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11416","pdf":"https://arxiv.org/pdf/2606.11416","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.11416"},"evidence":{"snippet":"We introduce MPC-Patch-Bench, a repository-level benchmark organised around two frameworks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.11416"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MPC-Patch-Bench evaluates LLM-based code repair on repository-level Secure Multi-Party Computation (MPC) software. It provides 205 verified instances with Fail-to-Pass/Pass-to-Pass tests and a verifier that checks cryptographic safety and numerical fidelity via differential testing and static analysis.","whyItMatters":"Existing benchmarks lack MPC-aware evaluation for repository-level code repair. MPC-Patch-Bench addresses security and numerical-fidelity gaps, offering a repeatable protocol for assessing LLM agents on real-world MPC tasks, with verification rejected up to 40% of functionally passing patches.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5f0c20089eb1b773a816e838d7ef44624ec30c955fb68286621f83a0b5ae8754"},"motivation":"Repository-level benchmarks for evaluating Large Language Model (LLM) code repair on Secure Multi-Party Computation (MPC) software do not yet exist, and directly transplanting general-purpose benchmarks such as SWE-bench fails on three structural fronts: (i) MPC repositories are dominated by generic Python infrastructure rather than cryptographic logic; (ii) high-value MPC fixes lack the standardized tests rigid ext…","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11416","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_mpdocbench-parse_f77fb798","familyId":"bmf_790789c36119","name":"MPDocBench-Parse","oneLine":"MPDocBench-Parse is a benchmark for multi-page document parsing, covering 433 manually annotated documents with 3,246 pages across 15 document types in English and Chinese. It evaluates content fidelity and logical structure, including text, table, formula recognition, reading order, and heading hierarchy recovery.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22100","pdf":"https://arxiv.org/pdf/2605.22100","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.22100"},"evidence":{"snippet":"To address these gaps, we propose MPDocBench-Parse, a benchmark for multi-page document parsing in real-world applications.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.22100"},"ranking":{},"description":"MPDocBench-Parse is a benchmark for multi-page document parsing, covering 433 manually annotated documents with 3,246 pages across 15 document types in English and Chinese. It evaluates content fidelity and logical structure, including text, table, formula recognition, reading order, and heading hierarchy recovery.","whyItMatters":"The benchmark addresses the lack of realistic multi-page document parsing evaluation, offering a more comprehensive protocol for assessing semantic continuity, hierarchical structure recovery, and visual content preservation, which is crucial for advancing document parsing systems in practical applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"71f565417af5043db56aedc917cbb88c3b877c75d07086ea7814f969a8b01ce2"},"motivation":"Document parsing converts visually rich documents into machine-readable structured representations, forming a crucial foundation for information systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.22100","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_mpie-bench_16361495","familyId":"bmf_e29db8d408ba","name":"MPIE-Bench","oneLine":"MPIE-Bench evaluates multi-person image editing models on tasks involving multiple named people in contact interactions such as embrace, carry, or grapple. The benchmark provides a 2,500-sample test set with 14 interaction categories and four contact densities, and scores outputs on six axes including anatomy and interaction via mesh reconstruction.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27616","pdf":"https://arxiv.org/pdf/2607.27616","project":null,"code":"https://github.com/AnnLin0628/mpie-bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.27616"},"evidence":{"snippet":"We introduce MPIE-Bench, a 2,500-sample benchmark of video-mined editing triplets spanning 405 scenes, 14 interaction categories, and four contact densities (C0-C3).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":39,"hfDailySubmittedAt":"2026-07-31T00:00:00.000Z","githubStars":7,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27616"},"ranking":{"90d":{"score":40,"rank":148,"coverage":0.7,"confidence":"Medium"}},"description":"MPIE-Bench evaluates multi-person image editing models on tasks involving multiple named people in contact interactions such as embrace, carry, or grapple. The benchmark provides a 2,500-sample test set with 14 interaction categories and four contact densities, and scores outputs on six axes including anatomy and interaction via mesh reconstruction.","whyItMatters":"Existing evaluations often overlook anatomical and geometric errors in multi-person editing, and VLM-based judges may rate such errors as acceptable. This benchmark provides a geometry-based scoring method that tracks human judgment more closely, enabling more reliable comparisons of editing models on this challenging task.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7611bdd34d173844a0f058dc43ffaa34f24a05879d32efb35e197280ec69d0df"},"motivation":"Text-to-image and personalized editing models now synthesize high-fidelity single-subject images with ease.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27616","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MPIE-Bench Team","organizationType":"benchmark-organization","sourceUrl":"https://github.com/AnnLin0628/mpie-bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mr-lidar_0b6f3704","familyId":"bmf_81c065acf37e","name":"MR-LiDAR","oneLine":"MR-LiDAR is a multi-resolution LiDAR benchmark for roadside perception. It includes point cloud data and annotations from 16-, 32-, 80-, and 128-beam LiDARs in identical scenarios, covering vehicles and vulnerable road users at various distances.","area":"Robotics & Embodied AI","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24777","pdf":"https://arxiv.org/pdf/2605.24777","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24777"},"evidence":{"snippet":"To address this gap, we present MR-LiDAR, a controlled multi-resolution LiDAR benchmark for roadside perception diagnostics.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24777"},"ranking":{},"description":"MR-LiDAR is a multi-resolution LiDAR benchmark for roadside perception. It includes point cloud data and annotations from 16-, 32-, 80-, and 128-beam LiDARs in identical scenarios, covering vehicles and vulnerable road users at various distances.","whyItMatters":"The benchmark targets the need for empirical comparison of LiDAR configurations to guide sensor selection in roadside perception systems, addressing cost and performance trade-offs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1f592aff16e110d7eea06f5cc9a12bc913daddb20cf5dd497fc1356f88a096bf"},"motivation":"LiDAR model selection is a critical issue in roadside sensing systems, as it directly determines both perception capability and deployment cost.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24777","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"general"},{"id":"lib_mrcr","familyId":"family_mrcr","name":"MRCR","oneLine":"Established benchmark family · Long Context.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2409.12640","pdf":null,"project":"https://github.com/google-deepmind/eval_hub/tree/master/eval_hub/mrcr_v2","code":"https://github.com/google-deepmind/eval_hub/tree/master/eval_hub/mrcr_v2","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_mrcr"},"ranking":{},"recordType":"family","aliases":["Multi-Round Co-reference Resolution"],"sourceAttribution":[{"role":"official-repository","url":"https://github.com/google-deepmind/eval_hub/tree/master/eval_hub/mrcr_v2"}],"adoptionRefs":["google-gemini25"],"modelReportReferences":[{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"mrcr","url":"https://llm-stats.com/benchmarks/mrcr","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","general"],"catalogModelCount":7,"catalogStarCount":0},{"id":"catalog_f840bd7cecb4ce31","familyId":"catalog_family_f840bd7cecb4ce31","name":"MRCR 128K (2-needle)","oneLine":"MRCR (Multi-Round Coreference Resolution) at 128K context length with 2 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 2 items to retrieve.","description":"MRCR (Multi-Round Coreference Resolution) at 128K context length with 2 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 2 items to retrieve.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mrcr-128k-(2-needle)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f840bd7cecb4ce31"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mrcr-128k-(2-needle)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mrcr-128k-(2-needle)","url":"https://llm-stats.com/benchmarks/mrcr-128k-(2-needle)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_786a93fb2cb37291","familyId":"catalog_family_786a93fb2cb37291","name":"MRCR 128K (4-needle)","oneLine":"MRCR (Multi-Round Coreference Resolution) at 128K context length with 4 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 4 items to retrieve.","description":"MRCR (Multi-Round Coreference Resolution) at 128K context length with 4 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 4 items to retrieve.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mrcr-128k-(4-needle)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_786a93fb2cb37291"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mrcr-128k-(4-needle)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mrcr-128k-(4-needle)","url":"https://llm-stats.com/benchmarks/mrcr-128k-(4-needle)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_ab936e5d1333318c","familyId":"catalog_family_ab936e5d1333318c","name":"MRCR 128K (8-needle)","oneLine":"MRCR (Multi-Round Coreference Resolution) at 128K context length with 8 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 8 items to retrieve.","description":"MRCR (Multi-Round Coreference Resolution) at 128K context length with 8 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 128K-token contexts with 8 items to retrieve.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mrcr-128k-(8-needle)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ab936e5d1333318c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mrcr-128k-(8-needle)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mrcr-128k-(8-needle)","url":"https://llm-stats.com/benchmarks/mrcr-128k-(8-needle)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","general"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_9b716240d72b7367","familyId":"catalog_family_9b716240d72b7367","name":"MRCR 1M","oneLine":"MRCR 1M is a variant of the Multi-Round Coreference Resolution benchmark designed for testing extremely long context capabilities with approximately 1 million tokens. It evaluates models' ability to maintain reasoning and attention across ultra-long conversations.","description":"MRCR 1M is a variant of the Multi-Round Coreference Resolution benchmark designed for testing extremely long context capabilities with approximately 1 million tokens. It evaluates models' ability to maintain reasoning and attention across ultra-long conversations.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Long Context","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9b716240d72b7367"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mrcr1m"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mrcr-1m"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mrcr1m","url":"https://benchlm.ai/benchmarks/mrcr1m","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"MRCR 1M","format":"Long-context retrieval MMR","tasks":"Million-token retrieval","successorKey":null},{"catalog":"llm-stats","sourceId":"mrcr-1m","url":"https://llm-stats.com/benchmarks/mrcr-1m","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","long context","general"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_12daadf376bddccc","familyId":"catalog_family_12daadf376bddccc","name":"MRCR 1M (pointwise)","oneLine":"MRCR 1M (pointwise) is a variant of the Multi-Round Coreference Resolution benchmark that uses pointwise evaluation for ultra-long contexts (~1M tokens). This version evaluates each response independently rather than comparatively, testing models' absolute performance on long-context reasoning tasks.","description":"MRCR 1M (pointwise) is a variant of the Multi-Round Coreference Resolution benchmark that uses pointwise evaluation for ultra-long contexts (~1M tokens). This version evaluates each response independently rather than comparatively, testing models' absolute performance on long-context reasoning tasks.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mrcr-1m-(pointwise)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_12daadf376bddccc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mrcr-1m-(pointwise)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mrcr-1m-(pointwise)","url":"https://llm-stats.com/benchmarks/mrcr-1m-(pointwise)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_d5877324a6222192","familyId":"catalog_family_d5877324a6222192","name":"MRCR 64K (2-needle)","oneLine":"MRCR (Multi-Round Coreference Resolution) at 64K context length with 2 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 2 items to retrieve.","description":"MRCR (Multi-Round Coreference Resolution) at 64K context length with 2 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 2 items to retrieve.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mrcr-64k-(2-needle)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d5877324a6222192"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mrcr-64k-(2-needle)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mrcr-64k-(2-needle)","url":"https://llm-stats.com/benchmarks/mrcr-64k-(2-needle)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_54fe47056a26c695","familyId":"catalog_family_54fe47056a26c695","name":"MRCR 64K (4-needle)","oneLine":"MRCR (Multi-Round Coreference Resolution) at 64K context length with 4 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 4 items to retrieve.","description":"MRCR (Multi-Round Coreference Resolution) at 64K context length with 4 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 4 items to retrieve.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mrcr-64k-(4-needle)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_54fe47056a26c695"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mrcr-64k-(4-needle)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mrcr-64k-(4-needle)","url":"https://llm-stats.com/benchmarks/mrcr-64k-(4-needle)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_bfb9e5abf18fcd7b","familyId":"catalog_family_bfb9e5abf18fcd7b","name":"MRCR 64K (8-needle)","oneLine":"MRCR (Multi-Round Coreference Resolution) at 64K context length with 8 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 8 items to retrieve.","description":"MRCR (Multi-Round Coreference Resolution) at 64K context length with 8 needles. Models must navigate long conversations to reproduce specific model outputs, testing attention and reasoning across 64K-token contexts with 8 items to retrieve.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mrcr-64k-(8-needle)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bfb9e5abf18fcd7b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mrcr-64k-(8-needle)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mrcr-64k-(8-needle)","url":"https://llm-stats.com/benchmarks/mrcr-64k-(8-needle)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_f375a5d25b1929e3","familyId":"catalog_family_f375a5d25b1929e3","name":"MRCR v2 (8-needle)","oneLine":"MRCR v2 (8-needle) is a variant of the Multi-Round Coreference Resolution benchmark that includes 8 needle items to retrieve from long contexts. This tests models' ability to simultaneously track and reason about multiple pieces of information across extended conversations.","description":"MRCR v2 (8-needle) is a variant of the Multi-Round Coreference Resolution benchmark that includes 8 needle items to retrieve from long contexts. This tests models' ability to simultaneously track and reason about multiple pieces of information across extended conversations.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mrcr-v2-(8-needle)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f375a5d25b1929e3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mrcr-v2-(8-needle)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mrcr-v2-(8-needle)","url":"https://llm-stats.com/benchmarks/mrcr-v2-(8-needle)","datasetSlug":"mrcr-v2","versionCount":5,"subsetCount":28,"rowCount":null,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":["long context","reasoning","general"],"catalogModelCount":23,"catalogStarCount":1,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_9e25967de7cdd8b0","familyId":"catalog_family_9e25967de7cdd8b0","name":"MRCR v2 (8-needle, 512K-1M)","oneLine":"MRCR v2 8-needle variant evaluated on contexts from 512K to 1M tokens.","description":"MRCR v2 8-needle variant evaluated on contexts from 512K to 1M tokens.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mrcr-v2-8-needle-512k-1m","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9e25967de7cdd8b0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mrcr-v2-8-needle-512k-1m"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mrcr-v2-8-needle-512k-1m","url":"https://llm-stats.com/benchmarks/mrcr-v2-8-needle-512k-1m","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","general"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_2cb15f6ff9750f45","familyId":"catalog_family_2cb15f6ff9750f45","name":"MRCR v2 128K-256K","oneLine":"MRCR v2 slice focused on very long contexts at 128K-256K lengths.","description":"MRCR v2 slice focused on very long contexts at 128K-256K lengths.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2cb15f6ff9750f45"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mrcr-v2-128k-256k"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mrcrv2_128_256","url":"https://benchlm.ai/benchmarks/mrcr-v2-128k-256k","paperUrl":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","year":"2026","fullName":"OpenAI MRCR v2 8-needle 128K-256K","format":"Very-long-context retrieval","tasks":"8-needle retrieval tasks","successorKey":null}],"catalogCategories":["reasoning"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_6724e55fa567213a","familyId":"catalog_family_6724e55fa567213a","name":"MRCR v2 64K-128K","oneLine":"MRCR v2 slice focused on long-context retrieval at 64K-128K lengths.","description":"MRCR v2 slice focused on long-context retrieval at 64K-128K lengths.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6724e55fa567213a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mrcr-v2-64k-128k"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mrcrv2_64_128","url":"https://benchlm.ai/benchmarks/mrcr-v2-64k-128k","paperUrl":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","year":"2026","fullName":"OpenAI MRCR v2 8-needle 64K-128K","format":"Long-context retrieval","tasks":"8-needle retrieval tasks","successorKey":null}],"catalogCategories":["reasoning"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_11d6a18b4f745fb6","familyId":"catalog_family_11d6a18b4f745fb6","name":"MRCRv2","oneLine":"MRCR v2 (Multi-Round Coreference Resolution version 2) is an enhanced version of the synthetic long-context reasoning task. It extends the original MRCR framework with improved evaluation criteria and additional complexity for testing models' ability to maintain attention and reasoning across extended contexts.","description":"MRCR v2 (Multi-Round Coreference Resolution version 2) is an enhanced version of the synthetic long-context reasoning task. It extends the original MRCR framework with improved evaluation criteria and additional complexity for testing models' ability to maintain attention and reasoning across extended contexts.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Long Context","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/introducing-gpt-5-2/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_11d6a18b4f745fb6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mrcrv2"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mrcr-v2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mrcrv2","url":"https://benchlm.ai/benchmarks/mrcrv2","paperUrl":"https://openai.com/index/introducing-gpt-5-2/","year":"2025","fullName":"MRCRv2","format":"Multi-round long-context evaluation","tasks":"Long-context retrieval","successorKey":null},{"catalog":"llm-stats","sourceId":"mrcr-v2","url":"https://llm-stats.com/benchmarks/mrcr-v2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","long context","general"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_mrmad_6774e059","familyId":"bmf_84c12f354a56","name":"MRMAD","oneLine":"Evaluates LALMs on audio degradation perception through multi-turn dialogues over multiple audio inputs, covering speech, music, and sound.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SD"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.22236v1","pdf":"https://arxiv.org/pdf/2608.22236v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce MRMAD, a Multi-Round Multi-Audio Degradation benchmark for evaluating audio degradation perception and understanding in LALMs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22236"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LALMs on audio degradation perception through multi-turn dialogues over multiple audio inputs, covering speech, music, and sound.","whyItMatters":"Audio degradation perception is critical for real-world audio understanding, yet existing benchmarks overlook this low-level diagnostic capability.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"ebf6bf227e9edf74bd8dd257e62370dff5ee94ab36ed085cbf5a6c7092871811"},"motivation":"Large audio-language models (LALMs) have shown promising progress in understanding speech, music, and general sound events, yet their ability to reason about how audio signals are degraded remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is clearly named, involves systematic evaluation of 18 models, and provides a public path via the arXiv paper.","canonicalNameSource":"paper_title","canonicalNameEvidence":"MRMAD: A Multi-Round Multi-Audio Benchmark for Evaluating Acoustic Degradation Perception in Large Audio-Language Models"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22236v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":35,"confidence":"Medium","horizon":"7d","reason":"Audio degradation perception is a specialized subfield that may attract only moderate attention from audio-language model researchers."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_ms-mlb_c7714a98","familyId":"bmf_23c6680cf29b","name":"MS-MLB","oneLine":"MS-MLB evaluates machine learning models for classifying multiple sclerosis versus healthy controls from whole blood RNA expression data (GSE17048). It uses a shared pipeline with nested cross-validation and a holdout set, and reports the MS Research Score composite metric.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.05196","pdf":"https://arxiv.org/pdf/2608.05196","project":null,"code":"https://github.com/duckyquang/MS-MLB","data":null,"hfPaper":"https://huggingface.co/papers/2608.05196"},"evidence":{"snippet":"This paper presents MS-MLB (Multiple Sclerosis Machine Learning Benchmark), a reproducible open benchmark for machine learning based MS research classification from whole blood RNA expression data.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.05196"},"ranking":{"30d":{"score":23,"rank":157,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":361,"coverage":0.55,"confidence":"Low"}},"description":"MS-MLB evaluates machine learning models for classifying multiple sclerosis versus healthy controls from whole blood RNA expression data (GSE17048). It uses a shared pipeline with nested cross-validation and a holdout set, and reports the MS Research Score composite metric.","whyItMatters":"Existing MS transcriptomic studies lack reproducible and standardized evaluation. This benchmark provides a leakage-controlled pipeline and an external model submission pathway for comparable research comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e481c1e43b4e5bd038667b3a3a00409fa49fd6f3600f95c41c72fc7d37c08531"},"motivation":"Multiple sclerosis (MS) is diagnosed through clinical assessment, magnetic resonance imaging, laboratory evidence when appropriate, and exclusion of better explanations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.05196","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Synthica Research Group","organizationType":"community","sourceUrl":"https://github.com/duckyquang/MS-MLB","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_msibench_1f439887","familyId":"bmf_0511d5a6a43f","name":"MSIBench","oneLine":"Evaluates multimodal large language models on discrimination, understanding, and reasoning under situational illusions.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-26","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.22232","pdf":"https://arxiv.org/pdf/2608.22232","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Building on this taxonomy, we introduce MSIBench, a benchmark designed to assess the discrimination, understanding, and reasoning capabilities of MLLMs under situational illusions.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22232"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates multimodal large language models on discrimination, understanding, and reasoning under situational illusions.","whyItMatters":"Tests whether MLLMs are misled when appearances deviate from physical states, addressing reliability in real-world perception.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"0f412802d806ce00f4d1fb4075b8484e6a6c4a057f31fa26a052e9e7138fbf7c"},"motivation":"Real-world situation appearances can deviate from their underlying physical states, challenging the reliability of multimodal large language models (MLLMs) in practical applications.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"No public release evidence beyond the paper; no code, data, or project link is supplied to verify a stable scoring contract or reuse path.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce MSIBench, a benchmark designed to assess the discrimination, understanding, and reasoning capabilities of MLLMs under situational illusions"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.22232","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":38,"confidence":"Low","horizon":"7d","reason":"The benchmark targets an emerging robustness issue for MLLMs, but the abstract alone offers limited signal for immediate wide adoption."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_msqa_623c5d9a","familyId":"bmf_96020ebf801d","name":"MSQA","oneLine":"MSQA evaluates 1,064 natively sourced questions across 11 language groups, five cultural dimensions, and three difficulty tiers, testing cultural knowledge in a multilingual context.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.00724","pdf":"https://arxiv.org/pdf/2607.00724","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.00724"},"evidence":{"snippet":"To test this assumption directly, we introduce MSQA, a benchmark of 1,064 natively sourced questions across 11 language groups, five cultural dimensions, and three difficulty tiers.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.00724"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MSQA evaluates 1,064 natively sourced questions across 11 language groups, five cultural dimensions, and three difficulty tiers, testing cultural knowledge in a multilingual context.","whyItMatters":"It addresses the gap in measuring cultural alignment separately from language ability, revealing that models often fail to achieve cultural competence despite multilingual fluency.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f48b33543b788537eecaaea8ef4a4a5e9a9b8975c58f636f3da28457f2ee0d44"},"motivation":"Multilingual fluency often invites a stronger assumption: a model that can speak a user's language must also understand the culture encoded by that language.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.00724","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"msqa","url":"https://llm-stats.com/benchmarks/msqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","general"],"catalogModelCount":2,"catalogStarCount":0},{"id":"catalog_0a7b4890e81878c3","familyId":"catalog_family_0a7b4890e81878c3","name":"MStar","oneLine":"A general visual question-answering benchmark used in provider tables for real-image reasoning quality.","description":"A general visual question-answering benchmark used in provider tables for real-image reasoning quality.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0a7b4890e81878c3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/mstar"}],"catalogSources":[{"catalog":"benchlm","sourceId":"mStar","url":"https://benchlm.ai/benchmarks/mstar","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"MStar","format":"Image-grounded QA","tasks":"Real-image visual QA","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0a207a8b12e78d0c","familyId":"catalog_family_0a207a8b12e78d0c","name":"MT-AIME 2025","oneLine":"MT-AIME 2025 is Cohere's internal multilingual translation of AIME 2025, evaluated for Arabic, Japanese, and Korean in the Command A+ release.","description":"MT-AIME 2025 is Cohere's internal multilingual translation of AIME 2025, evaluated for Arabic, Japanese, and Korean in the Command A+ release.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mt-aime-2025","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0a207a8b12e78d0c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mt-aime-2025"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mt-aime-2025","url":"https://llm-stats.com/benchmarks/mt-aime-2025","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","math","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_fa3c844fc8547162","familyId":"catalog_family_fa3c844fc8547162","name":"MT-Bench","oneLine":"MT-Bench is a challenging multi-turn benchmark that measures the ability of large language models to engage in coherent, informative, and engaging conversations. It uses strong LLMs as judges for scalable and explainable evaluation of multi-turn dialogue capabilities.","description":"MT-Bench is a challenging multi-turn benchmark that measures the ability of large language models to engage in coherent, informative, and engaging conversations. It uses strong LLMs as judges for scalable and explainable evaluation of multi-turn dialogue capabilities.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Roleplay","General","Communication","Creativity"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mt-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fa3c844fc8547162"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mt-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mt-bench","url":"https://llm-stats.com/benchmarks/mt-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","roleplay","general","communication","creativity"],"catalogModelCount":12,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_mt-web2code_5f98b4bc","familyId":"bmf_19c14f7a2f67","name":"MT-Web2Code","oneLine":"MT-Web2Code evaluates coding agents on multi-turn web UI reconstruction and modification tasks across 102 tasks in 16 domains, with a dual-axis protocol measuring target-region fidelity and preservation of unaffected content.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03474","pdf":"https://arxiv.org/pdf/2608.03474","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03474"},"evidence":{"snippet":"To bridge this gap, we introduce MT-Web2Code, the first multimodal coding benchmark for multi-turn Macro-Level Regional Reconstruction and Micro-Level Localized Modification, which contains 102 tasks spanning 16 vertical domains.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03474"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MT-Web2Code evaluates coding agents on multi-turn web UI reconstruction and modification tasks across 102 tasks in 16 domains, with a dual-axis protocol measuring target-region fidelity and preservation of unaffected content.","whyItMatters":"Existing benchmarks focus on single-turn full-page generation, missing the iterative workflow of real frontend engineering. This benchmark aims to fill that gap and identify weaknesses in multi-turn UI coding agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e1f5c6f94f724e97ea277824df0b568dc710f1b4baedf7f7cc40b530bca4a3a7"},"motivation":"Recent advances in Large Vision-Language Models (LVLMs) have demonstrated impressive capabilities in web UI generation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03474","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mtavg-bench_0372b3a9","familyId":"bmf_ac59d735d593","name":"MTAVG-Bench","oneLine":"MTAVG-Bench 2.0 evaluates omni large language models on diagnosing high-level cinematic failures in multi-talker audio-video generation, with over 10,000 QA instances covering acting, narrative, atmosphere, and audio-visual language.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.28035","pdf":"https://arxiv.org/pdf/2605.28035","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28035"},"evidence":{"snippet":"To fill this gap, we introduce MTAVG-Bench 2.0, a benchmark for diagnosing failure modes of cinematic expressiveness in multi-talker audio-video generation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28035"},"ranking":{},"description":"MTAVG-Bench 2.0 evaluates omni large language models on diagnosing high-level cinematic failures in multi-talker audio-video generation, with over 10,000 QA instances covering acting, narrative, atmosphere, and audio-visual language.","whyItMatters":"Standard metrics like lip-sync do not capture cinematic expressiveness; this benchmark targets a gap in evaluating higher-level audio-visual quality in scene-level generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0dbaa6e642f474af1f621756a2d6f20f876f9c4ea1b54424b218d5de7b88e490"},"motivation":"In recent years, Multi-Talker Audio-Video Generation (MTAVG) models have shown promising performance on fundamental metrics such as lip-sync and audio-visual alignment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28035","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mteb-br_cc2fe8a1","familyId":"bmf_16304415cf38","name":"MTEB-BR","oneLine":"MTEB-BR is a text embedding benchmark for Brazilian Portuguese with 22 native tasks across seven categories, including classification, clustering, retrieval, and reranking. It evaluates models on native Portuguese data and provides a public leaderboard.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.04581","pdf":"https://arxiv.org/pdf/2607.04581","project":"https://doi.org/10.5281/zenodo.21087216","code":"https://github.com/tardellirs/mteb-br","data":null,"hfPaper":"https://huggingface.co/papers/2607.04581"},"evidence":{"snippet":"We introduce MTEB-BR, a benchmark of 22 native Brazilian-Portuguese tasks across seven categories (classification, multilabel classification, pair classification, semantic textual similarity, clustering, retrieval, and reranking), admitting only data created or found in Portuguese and excluding translations by construction.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":11,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.04581"},"ranking":{"90d":{"score":34,"rank":206,"coverage":0.7,"confidence":"Medium"}},"description":"MTEB-BR is a text embedding benchmark for Brazilian Portuguese with 22 native tasks across seven categories, including classification, clustering, retrieval, and reranking. It evaluates models on native Portuguese data and provides a public leaderboard.","whyItMatters":"Portuguese embedding evaluation lacked native benchmarks. This benchmark provides statistically grounded model comparison, showing that multilingual leaderboards only moderately predict Portuguese performance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1c20f055685d32699e3ebefb3a09fa498ff4594988aebef3557c3af32a27ccb8"},"motivation":"Text embeddings for Portuguese have no dedicated benchmark: evaluation rests on translated corpora such as English MS MARCO or on thin multilingual coverage, with native tasks scattered and unconsolidated.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.04581","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_mtm-bench_e1a4ed71","familyId":"bmf_a0d4dfc5a508","name":"MTM-Bench","oneLine":"MTM-Bench is a controlled benchmark for language-conditioned task execution in multilingual settings, enumerating all 27 instruction-content-response language triplets across English, Spanish, and Chinese. It contains 2,430 instances per model across semantic reversal, final-state extraction, and language purity tasks, with decomposed metrics for semantic correctness, language adherence, constraint satisfaction, and joint success.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27649","pdf":"https://arxiv.org/pdf/2605.27649","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27649"},"evidence":{"snippet":"We introduce MTM-Bench, a controlled benchmark for language-conditioned task execution in which each instance is defined by a triplet \\((L_{\\text{instr}}, L_{\\text{content}}, L_{\\text{resp}})\\).","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27649"},"ranking":{},"description":"MTM-Bench is a controlled benchmark for language-conditioned task execution in multilingual settings, enumerating all 27 instruction-content-response language triplets across English, Spanish, and Chinese. It contains 2,430 instances per model across semantic reversal, final-state extraction, and language purity tasks, with decomposed metrics for semantic correctness, language adherence, constraint satisfaction, and joint success.","whyItMatters":"Multilingual LLMs are used when instruction, source content, and response languages differ, yet existing evaluations rarely isolate these roles. MTM-Bench provides a fully crossed design to attribute degradation to specific language roles, revealing that response-slot mismatch drives most performance loss and that mismatch count is not a monotonic predictor of difficulty.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"753bd38a7699932ff39b720a13aa24419a9b3088915cdaae11a259e0b053709b"},"motivation":"Multilingual LLMs are increasingly used when instruction, source content, and required response languages do not coincide.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27649","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_6aef73a46f67bebd","familyId":"catalog_family_6aef73a46f67bebd","name":"MTVQA","oneLine":"MTVQA (Multilingual Text-Centric Visual Question Answering) is the first benchmark featuring high-quality human expert annotations across 9 diverse languages, consisting of 6,778 question-answer pairs across 2,116 images. It addresses visual-textual misalignment problems in multilingual text-centric VQA.","description":"MTVQA (Multilingual Text-Centric Visual Question Answering) is the first benchmark featuring high-quality human expert annotations across 9 diverse languages, consisting of 6,778 question-answer pairs across 2,116 images. It addresses visual-textual misalignment problems in multilingual text-centric VQA.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Text-To-Image","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/mtvqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6aef73a46f67bebd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/mtvqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"mtvqa","url":"https://llm-stats.com/benchmarks/mtvqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","text-to-image","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_743371fe4e9f7af0","familyId":"catalog_family_743371fe4e9f7af0","name":"MuirBench","oneLine":"A comprehensive benchmark for robust multi-image understanding capabilities of multimodal LLMs. Consists of 12 diverse multi-image tasks involving 10 categories of multi-image relations (e.g., multiview, temporal relations, narrative, complementary). Comprises 11,264 images and 2,600 multiple-choice questions created in a pairwise manner, where each standard instance is paired with an unanswerable variant for reliable assessment.","description":"A comprehensive benchmark for robust multi-image understanding capabilities of multimodal LLMs. Consists of 12 diverse multi-image tasks involving 10 categories of multi-image relations (e.g., multiview, temporal relations, narrative, complementary). Comprises 11,264 images and 2,600 multiple-choice questions created in a pairwise manner, where each standard instance is paired with an unanswerable variant for reliable assessment.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/muirbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_743371fe4e9f7af0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/muirbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"muirbench","url":"https://llm-stats.com/benchmarks/muirbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":12,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_mulrobbench_1e33deff","familyId":"bmf_65e0c7a5af31","name":"MulRobBench","oneLine":"An offline protocol-conditioned benchmark for Vision-Language-Action UAV agents, evaluating operational context understanding, multimodal evidence arbitration, degradation-aware reasoning, and risk-aware action planning across 3,024 samples with semantic scoring and structural diagnostics.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.23870","pdf":"https://arxiv.org/pdf/2607.23870","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.23870"},"evidence":{"snippet":"We introduce MulRobBench, an offline, protocol-conditioned benchmark for Vision-Language-Action (VLA) UAV agents in smart-city environments.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23870"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"An offline protocol-conditioned benchmark for Vision-Language-Action UAV agents, evaluating operational context understanding, multimodal evidence arbitration, degradation-aware reasoning, and risk-aware action planning across 3,024 samples with semantic scoring and structural diagnostics.","whyItMatters":"Most UAV benchmarks focus on perception or navigation, leaving a gap in assessing coupled physical evidence, protocol constraints, and action risk. MulRobBench's diagnostic dimensions could inform safety-critical UAV deployment decisions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7682d3bd8cd4ab6557a48eb7c2a9eca8ef94a2f94d0283ca218d970cdb977320"},"motivation":"Smart-city airspace is transforming Uncrewed Aerial Vehicles (UAVs) from passive sensing platforms into cyber-physical decision makers that must follow operational rules under degraded observations and ambiguous language.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.23870","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_91a0d15dc744c7b3","familyId":"catalog_family_91a0d15dc744c7b3","name":"Multi-Challenge","oneLine":"MultiChallenge is a realistic multi-turn conversation evaluation benchmark that challenges frontier LLMs across four key categories: instruction retention (maintaining instructions throughout conversations), inference memory (recalling and connecting details from previous turns), reliable versioned editing (adapting to evolving instructions during collaborative editing), and self-coherence (avoiding contradictions in responses). The benchmark evaluates models on sustained, contextually complex dialogues across diverse topics including travel planning, technical documentation, and professional communication.","description":"MultiChallenge is a realistic multi-turn conversation evaluation benchmark that challenges frontier LLMs across four key categories: instruction retention (maintaining instructions throughout conversations), inference memory (recalling and connecting details from previous turns), reliable versioned editing (adapting to evolving instructions during collaborative editing), and self-coherence (avoiding contradictions in responses). The benchmark evaluates models on sustained, contextually complex dialogues across diverse topics including travel planning, technical documentation, and professional communication.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Communication"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/multichallenge","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_91a0d15dc744c7b3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/multichallenge"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"multichallenge","url":"https://llm-stats.com/benchmarks/multichallenge","datasetSlug":"multichallenge","versionCount":6,"subsetCount":5,"rowCount":null,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":["reasoning","communication"],"catalogModelCount":29,"catalogStarCount":1,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_56756bc009f4e6ce","familyId":"catalog_family_56756bc009f4e6ce","name":"Multi-IF","oneLine":"Multi-IF benchmarks LLMs on multi-turn and multilingual instruction following. It expands upon IFEval by incorporating multi-turn sequences and translating English prompts into 7 other languages, resulting in 4,501 multilingual conversations with three turns each. The benchmark reveals that current leading LLMs struggle with maintaining accuracy in multi-turn instructions and shows higher error rates for non-Latin script languages.","description":"Multi-IF benchmarks LLMs on multi-turn and multilingual instruction following. It expands upon IFEval by incorporating multi-turn sequences and translating English prompts into 7 other languages, resulting in 4,501 multilingual conversations with three turns each. The benchmark reveals that current leading LLMs struggle with maintaining accuracy in multi-turn instructions and shows higher error rates for non-Latin script languages.","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Instruction Following","Language","Reasoning","Structured Output","Communication"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/multi-if","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_56756bc009f4e6ce"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/multi-if"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"multi-if","url":"https://llm-stats.com/benchmarks/multi-if","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["instruction following","language","reasoning","structured output","communication"],"catalogModelCount":23,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general"},{"id":"bm_multi-lcb_028b5b88","familyId":"bmf_d210c1214987","name":"Multi-LCB","oneLine":"Multi-LCB evaluates code generation across twelve programming languages (C++, C#, Python, Java, Rust, Go, TypeScript, JavaScript, Ruby, Kotlin, Scala, PHP) by transforming Python tasks from LiveCodeBench while preserving its contamination controls and evaluation protocol.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation"],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.20517","pdf":"https://arxiv.org/pdf/2606.20517","project":null,"code":"https://github.com/Multi-LCB/Multi-LCB","data":null,"hfPaper":"https://huggingface.co/papers/2606.20517"},"evidence":{"snippet":"We introduce Multi-LCB, a benchmark for evaluating LLMs across twelve programming languages, including Python.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":60,"hfDailySubmittedAt":"2026-06-19T00:00:00.000Z","githubStars":28,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.20517"},"ranking":{"90d":{"score":51,"rank":57,"coverage":0.7,"confidence":"Medium"}},"description":"Multi-LCB evaluates code generation across twelve programming languages (C++, C#, Python, Java, Rust, Go, TypeScript, JavaScript, Ruby, Kotlin, Scala, PHP) by transforming Python tasks from LiveCodeBench while preserving its contamination controls and evaluation protocol.","whyItMatters":"LiveCodeBench restricted code evaluation to Python; Multi-LCB addresses the gap by enabling cross-language assessment, revealing Python overfitting and language-specific contamination in LLMs, and supporting robust multilingual code evaluation for real-world software engineering.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f648ab30e00c8861c6e20a2507217f1d4853a08596d77d744506b2efe8608d6f"},"motivation":"LiveCodeBench (LCB) has recently become a widely adopted benchmark for evaluating large language models (LLMs) on code-generation tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.20517","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Multi-LCB Team","organizationType":"academic-lab","sourceUrl":"https://github.com/Multi-LCB/Multi-LCB","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_multi-legal-bench_2cfb13a9","familyId":"bmf_608eaa134d4c","name":"Multi-Legal-Bench","oneLine":"Multi-Legal-Bench evaluates LLMs on legal reasoning across six countries, four language families, and five tasks, using structured metadata from court registries for classification and extraction.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Inspectable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29738","pdf":"https://arxiv.org/pdf/2605.29738","project":null,"code":null,"data":"https://huggingface.co/datasets/overthelex/multi-legal-bench","hfPaper":"https://huggingface.co/papers/2605.29738"},"evidence":{"snippet":"We introduce Multi-Legal-Bench, the first cross-jurisdictional legal benchmark that evaluates identical tasks across six countries (Ukraine, France, Netherlands, Poland, Czech Republic, Lithuania), four language families, and 165 million full-text court decisions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":109,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2605.29738"},"ranking":{},"description":"Multi-Legal-Bench evaluates LLMs on legal reasoning across six countries, four language families, and five tasks, using structured metadata from court registries for classification and extraction.","whyItMatters":"It enables cross-lingual and cross-jurisdictional comparison of legal reasoning, revealing that transfer quality depends more on label-set alignment than language proximity, aiding model selection in legal NLP.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"932986e3935f9b1ed96a88e924160e52fb91ce946cbdbcbc32308527a3625f84"},"motivation":"Legal NLP benchmarks overwhelmingly evaluate a single language or aggregate tasks that differ fundamentally across jurisdictions, making cross-lingual comparison impossible.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29738","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_ct-segmentation-benchmark_d7d89501","familyId":"bmf_2a58534eb608","name":"Multi-Organ CT Segmentation Benchmark","oneLine":"Compares MOOSE, TotalSegmentator, and VoxTell on public CT datasets using Dice and IoU scores per organ, with a staged design from a 4-organ pilot to 16 organs across up to 50 subjects.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Software & Cloud","Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/caitlin-leonard/ct-segmentation-benchmark","pdf":null,"project":"https://drive.google.com/file/d/1jloIo4DEvmwkRSWtV7dEShO3bL9ReBie/view?usp=sharing","code":"https://github.com/caitlin-leonard/ct-segmentation-benchmark","data":"https://zenodo.org/record/6802614","hfPaper":null},"evidence":{"snippet":"ct-segmentation-benchmark Staged Dice/IoU benchmark of MOOSE, TotalSegmentator & VoxTell on public CT datasets.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:caitlin-leonard/ct-segmentation-benchmark"},"ranking":{"30d":{"score":23,"rank":125,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":329,"coverage":0.55,"confidence":"Low"}},"description":"Compares MOOSE, TotalSegmentator, and VoxTell on public CT datasets using Dice and IoU scores per organ, with a staged design from a 4-organ pilot to 16 organs across up to 50 subjects.","whyItMatters":"Demonstrates that tool rankings in medical image segmentation depend on sample size and organ set, providing a reproducible protocol for fair comparison of public segmentation tools.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"cb48bf518338408c772925d4827ab1c6501edb9e25d8a95b157f76da6f2f34d6"},"motivation":"ct-segmentation-benchmark Staged Dice/IoU benchmark of MOOSE, TotalSegmentator & VoxTell on public CT datasets.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/caitlin-leonard/ct-segmentation-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":35,"confidence":"Low","horizon":"7d","reason":"The benchmark addresses a practical medical imaging comparison but lacks a high-profile venue or community driver, suggesting moderate niche interest."},"evaluationMode":"public_reusable","publishers":[{"name":"Caitlin Leonard","organizationType":"academic-lab","sourceUrl":"https://github.com/caitlin-leonard/ct-segmentation-benchmark","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Multimodal Perception","Coding & Software Engineering"],"domainScope":"specific"},{"id":"catalog_37f7f4cd7dee3189","familyId":"catalog_family_37f7f4cd7dee3189","name":"Multi-SWE Bench","oneLine":"A multilingual benchmark for issue resolving that evaluates Large Language Models' ability to resolve software issues across diverse programming ecosystems. Covers 7 programming languages (Java, TypeScript, JavaScript, Go, Rust, C, and C++) with 1,632 high-quality instances carefully annotated by 68 expert annotators. Addresses limitations of existing benchmarks that focus almost exclusively on Python.","description":"A multilingual benchmark for issue resolving that evaluates Large Language Models' ability to resolve software issues across diverse programming ecosystems. Covers 7 programming languages (Java, TypeScript, JavaScript, Go, Rust, C, and C++) with 1,632 high-quality instances carefully annotated by 68 expert annotators. Addresses limitations of existing benchmarks that focus almost exclusively on Python.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.minimax.io/news/minimax-m27-en","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_37f7f4cd7dee3189"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/multiswebench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/multi-swe-bench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"multiSweBench","url":"https://benchlm.ai/benchmarks/multiswebench","paperUrl":"https://www.minimax.io/news/minimax-m27-en","year":"2026","fullName":"Multi-SWE Bench","format":"Repository task completion","tasks":"Multi-language repo tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"multi-swe-bench","url":"https://llm-stats.com/benchmarks/multi-swe-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","reasoning","code"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_multi2av-safety_13dc6c17","familyId":"bmf_49c9d04bc162","name":"Multi2AV-Safety","oneLine":"Audio-video generation is rapidly moving from prompt-driven synthesis toward multimodal conditioning, where text, images, audio, and video can jointly shape the generated output.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["Multimodal","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.26535","pdf":"https://arxiv.org/pdf/2608.26535","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.26535"},"evidence":{"snippet":"To bridge this gap, we introduce Multi2AV-Safety, the first safety benchmark, to the best of our knowledge, to cover all 11 non-singleton T/I/A/V conditioning configurations for audio-video generation, comprising 11,024 attack instances.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26535"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"Audio-video generation is rapidly moving from prompt-driven synthesis toward multimodal conditioning, where text, images, audio, and video can jointly shape the generated output.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.26535","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_multiglobeqa_73949294","familyId":"bmf_6185f11d15a8","name":"MultiGlobeQA","oneLine":"MultiGlobeQA evaluates geospatial reasoning in large language models across 46,060 question-answer pairs in 17 languages, covering 14 spatial-function families and 15 answer formats, with ground truth from three knowledge graphs.","area":"Language & Knowledge","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Logistics"],"capabilities":["Reasoning","Factuality"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03882","pdf":"https://arxiv.org/pdf/2608.03882","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03882"},"evidence":{"snippet":"We introduce MultiGlobeQA, a multilingual benchmark of 46,060 question-answer pairs spanning 14 spatial-function families and 15 answer formats, with execution-based ground truth over three knowledge graphs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03882"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"MultiGlobeQA evaluates geospatial reasoning in large language models across 46,060 question-answer pairs in 17 languages, covering 14 spatial-function families and 15 answer formats, with ground truth from three knowledge graphs.","whyItMatters":"Current geospatial benchmarks are limited in language coverage and geographic control. This benchmark aims to provide a broader, multilingual evaluation to identify specific failures in computational reasoning over geographic knowledge.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c80cca706fd7ab8e85516160919e62f76cce7721f812ffcf4b00855bed45ae84"},"motivation":"Geospatial reasoning, i.e., computing distances, containment, and other spatial relations over real-world entities, is central to navigation and logistics, yet large language models (LLMs) struggle with the required geometric and topological computation despite storing considerable geographic knowledge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03882","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_6ae40f22c77bfae7","familyId":"catalog_family_6ae40f22c77bfae7","name":"MultiLF","oneLine":"MultiLF benchmark","description":"MultiLF benchmark","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/multilf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6ae40f22c77bfae7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/multilf"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"multilf","url":"https://llm-stats.com/benchmarks/multilf","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["general"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d729b8fc6431d5a5","familyId":"catalog_family_d729b8fc6431d5a5","name":"Multilingual MGSM (CoT)","oneLine":"Multilingual Grade School Math (MGSM) benchmark evaluates language models' chain-of-thought reasoning abilities across ten typologically diverse languages. Contains 250 grade-school math problems manually translated from GSM8K dataset into languages including Bengali and Swahili.","description":"Multilingual Grade School Math (MGSM) benchmark evaluates language models' chain-of-thought reasoning abilities across ten typologically diverse languages. Contains 250 grade-school math problems manually translated from GSM8K dataset into languages including Bengali and Swahili.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/multilingual-mgsm-(cot)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d729b8fc6431d5a5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/multilingual-mgsm-(cot)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"multilingual-mgsm-(cot)","url":"https://llm-stats.com/benchmarks/multilingual-mgsm-(cot)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_6cf53613aa96f8f6","familyId":"catalog_family_6cf53613aa96f8f6","name":"Multilingual MMLU","oneLine":"MMLU-ProX is a comprehensive multilingual benchmark covering 29 typologically diverse languages, building upon MMLU-Pro. Each language version consists of 11,829 identical questions enabling direct cross-linguistic comparisons. The benchmark evaluates large language models' reasoning capabilities across linguistic and cultural boundaries through challenging, reasoning-focused questions with 10 answer choices.","description":"MMLU-ProX is a comprehensive multilingual benchmark covering 29 typologically diverse languages, building upon MMLU-Pro. Each language version consists of 11,829 identical questions enabling direct cross-linguistic comparisons. The benchmark evaluates large language models' reasoning capabilities across linguistic and cultural boundaries through challenging, reasoning-focused questions with 10 answer choices.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/multilingual-mmlu","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6cf53613aa96f8f6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/multilingual-mmlu"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"multilingual-mmlu","url":"https://llm-stats.com/benchmarks/multilingual-mmlu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning","general"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_0ad9d08786c3ce65","familyId":"catalog_family_0ad9d08786c3ce65","name":"MultiLoKo","oneLine":"A multilingual/localized knowledge benchmark reported in DeepSeek-V4 base-model evaluations.","description":"A multilingual/localized knowledge benchmark reported in DeepSeek-V4 base-model evaluations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0ad9d08786c3ce65"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/multiloko"}],"catalogSources":[{"catalog":"benchlm","sourceId":"multiLoKo","url":"https://benchlm.ai/benchmarks/multiloko","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"MultiLoKo","format":"Exact match","tasks":"Localized multilingual knowledge questions","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_3f8a7e66329a6d6c","familyId":"catalog_family_3f8a7e66329a6d6c","name":"MultiPL-E","oneLine":"MultiPL-E is a scalable and extensible system for translating unit test-driven code generation benchmarks to multiple programming languages. It extends HumanEval and MBPP Python benchmarks to 18 additional programming languages, enabling evaluation of neural code generation models across diverse programming paradigms and language features.","description":"MultiPL-E is a scalable and extensible system for translating unit test-driven code generation benchmarks to multiple programming languages. It extends HumanEval and MBPP Python benchmarks to 18 additional programming languages, enabling evaluation of neural code generation models across diverse programming paradigms and language features.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/multipl-e","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3f8a7e66329a6d6c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/multipl-e"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"multipl-e","url":"https://llm-stats.com/benchmarks/multipl-e","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","general"],"catalogModelCount":13,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_8eda1f657fbd4e93","familyId":"catalog_family_8eda1f657fbd4e93","name":"Multipl-E HumanEval","oneLine":"MultiPL-E is a scalable and extensible approach to benchmarking neural code generation that translates unit test-driven code generation benchmarks across multiple programming languages. It extends the HumanEval benchmark to 18 additional programming languages, enabling evaluation of code generation models across diverse programming paradigms and providing insights into how models generalize programming knowledge across language boundaries.","description":"MultiPL-E is a scalable and extensible approach to benchmarking neural code generation that translates unit test-driven code generation benchmarks across multiple programming languages. It extends the HumanEval benchmark to 18 additional programming languages, enabling evaluation of code generation models across diverse programming paradigms and providing insights into how models generalize programming knowledge across language boundaries.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/multipl-e-humaneval","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8eda1f657fbd4e93"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/multipl-e-humaneval"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"multipl-e-humaneval","url":"https://llm-stats.com/benchmarks/multipl-e-humaneval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","general"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_43a27979a119e9f8","familyId":"catalog_family_43a27979a119e9f8","name":"Multipl-E MBPP","oneLine":"MultiPL-E extends the Mostly Basic Python Problems (MBPP) benchmark to 18+ programming languages for evaluating multilingual code generation capabilities. MBPP contains 974 crowd-sourced programming problems designed to be solvable by entry-level programmers, covering programming fundamentals and standard library functionality. Each problem includes a task description, code solution, and automated test cases.","description":"MultiPL-E extends the Mostly Basic Python Problems (MBPP) benchmark to 18+ programming languages for evaluating multilingual code generation capabilities. MBPP contains 974 crowd-sourced programming problems designed to be solvable by entry-level programmers, covering programming fundamentals and standard library functionality. Each problem includes a task description, code solution, and automated test cases.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/multipl-e-mbpp","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_43a27979a119e9f8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/multipl-e-mbpp"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"multipl-e-mbpp","url":"https://llm-stats.com/benchmarks/multipl-e-mbpp","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_multiref-compass_396a4597","familyId":"bmf_83d3a7c119e7","name":"MultiRef-Compass","oneLine":"MultiRef-Compass evaluates multi-reference-to-audio-video generation systems on 350 curated samples. It assesses Basic Quality, Reference Consistency, Audio-Visual Consistency, and Instruction Following using 14 sub-metrics, combining automatic metrics with an MLLM-as-a-Judge framework.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14189","pdf":"https://arxiv.org/pdf/2607.14189","project":null,"code":"https://github.com/zxhhh0201/MultiRef-Compass","data":null,"hfPaper":"https://huggingface.co/papers/2607.14189"},"evidence":{"snippet":"To address this gap, we introduce MultiRef-Compass, a unified benchmark for MR2AV generation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":34,"hfDailySubmittedAt":null,"githubStars":12,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14189"},"ranking":{"90d":{"score":44,"rank":121,"coverage":0.7,"confidence":"Medium"}},"description":"MultiRef-Compass evaluates multi-reference-to-audio-video generation systems on 350 curated samples. It assesses Basic Quality, Reference Consistency, Audio-Visual Consistency, and Instruction Following using 14 sub-metrics, combining automatic metrics with an MLLM-as-a-Judge framework.","whyItMatters":"Existing benchmarks focus on text-driven or single-reference generation and often ignore joint audio-video alignment. MultiRef-Compass addresses the gap by providing a public protocol for multi-reference composition, enabling reproducible comparison across models in a rapidly evolving generation task.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"56752da5440a76ba771efb3eed9a915e47f283e2c763da850ea795095b8f28a4"},"motivation":"Multi-reference-to-audio-video (MR2AV) generation aims to generate coherent audio-video content conditioned on multiple references and textual instructions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14189","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MultiRef-Compass Team","organizationType":"academic-lab","sourceUrl":"https://github.com/zxhhh0201/MultiRef-Compass","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_multiseismo_9617a097","familyId":"bmf_e1646a92b259","name":"MULTISEISMO","oneLine":"MultiSeismo is a multimodal seismic dataset with over 16K events integrating waveforms, intensity maps, population exposure, and text, plus MISCE instruction set for seismic reasoning tasks like retrieval and cross-modal analysis.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.26320","pdf":"https://arxiv.org/pdf/2605.26320","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.26320"},"evidence":{"snippet":"These results prove that MultiSeismo provides a rigorous benchmark for future multimodal research in seismology and validate the success of our domain specific architectural adaptations.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26320"},"ranking":{},"description":"MultiSeismo is a multimodal seismic dataset with over 16K events integrating waveforms, intensity maps, population exposure, and text, plus MISCE instruction set for seismic reasoning tasks like retrieval and cross-modal analysis.","whyItMatters":"It enables evaluation of multimodal models on specialized scientific data, highlighting challenges in time-series processing and supporting development of domain-specific models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cb486fc30dd6a8f9883f28d18e07c78765e68ae64ff6bdd71a5944140c5ceaf7"},"motivation":"The application of generalist multimodal models (GMMs) to specialized scientific domains remains limited due to the scarcity of comprehensive domain-specific datasets that integrate multiple data modalities beyond text and images.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26320","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_multivationbench_9e6dc7dd","familyId":"bmf_7a98442d82f1","name":"MultivationBench","oneLine":"Evaluates multimodal motivation reasoning in story-driven visual narratives, using 1,000 narratives with 16,092 multi-label questions grounded in Maslow and Reiss frameworks across definition and practical reasoning tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.26465","pdf":"https://arxiv.org/pdf/2607.26465","project":null,"code":"https://github.com/HKUST-KnowComp/MultivationBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.26465"},"evidence":{"snippet":"To address this gap, we introduce MultivationBench, a benchmark designed to rigorously evaluate multimodal motivation reasoning within story-driven visual narratives.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26465"},"ranking":{"90d":{"score":28,"rank":270,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates multimodal motivation reasoning in story-driven visual narratives, using 1,000 narratives with 16,092 multi-label questions grounded in Maslow and Reiss frameworks across definition and practical reasoning tasks.","whyItMatters":"Assesses whether models can reason about evolving character motivations across sequential context, an underexplored capability for social intelligence.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"19bb77e91f48cd998c6b16c6cafab07a4b4d419e86fbe7ef854ff7182322a875"},"motivation":"Multimodal Large Language Models have sparked significant interest due to their potential for social intelligence; however, their ability to perform sequential motivation reasoning remains insufficiently studied.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26465","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_multiview-bench_66d1791f","familyId":"bmf_6e0786a96b06","name":"MultiView-Bench","oneLine":"Evaluates multi-view integration in vision-language models using diagnostic tasks for 3D scene comprehension, with a fixed-view baseline and proposed ViewNavigator method.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.08970","pdf":"https://arxiv.org/pdf/2607.08970","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.08970"},"evidence":{"snippet":"We introduce MultiView-Bench, a diagnostic benchmark expressly designed to evaluate multi-view integration for holistic 3D scene comprehension.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.08970"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates multi-view integration in vision-language models using diagnostic tasks for 3D scene comprehension, with a fixed-view baseline and proposed ViewNavigator method.","whyItMatters":"Addresses the gap in evaluating VLMs' ability to integrate observations across viewpoints into allocentric 3D models, a prerequisite for downstream tasks like assembly.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c5f573d94c580572008ee677d38da90719740e72a7bce3aa91d292e048c5ef15"},"motivation":"Recent benchmarks for VLMs largely assess single- or limited-view perception, leaving untested the core cognitive ability to integrate observations across viewpoints into a coherent, world-centric (allocentric) 3D mental model.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.08970","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_muse_8c1c770b","familyId":"bmf_4016c3db3bc3","name":"MUSE","oneLine":"MUSE evaluates text-to-CAD generation of complex B-Rep assemblies via design specifications. It scores models on code validity, geometric correctness, and design-intent alignment using rubrics covering functionality, manufacturability, and assemblability. A VLM judge with human validation is used for scalable scoring.","area":"Science & Engineering","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["CAD"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.28579","pdf":"https://arxiv.org/pdf/2605.28579","project":"https://dong7313.github.io/muse-benchmark/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28579"},"evidence":{"snippet":"To address this gap, we introduce MUSE, a Text-to-CAD benchmark focused on complex, editable boundary representation (B-Rep) assemblies.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28579"},"ranking":{},"description":"MUSE evaluates text-to-CAD generation of complex B-Rep assemblies via design specifications. It scores models on code validity, geometric correctness, and design-intent alignment using rubrics covering functionality, manufacturability, and assemblability. A VLM judge with human validation is used for scalable scoring.","whyItMatters":"Existing CAD benchmarks focus on single-part geometric similarity, missing industrial requirements. MUSE provides a structured evaluation that measures practical design quality, enabling progress toward engineering-ready CAD generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0397b51cc33fa03c4e4e04f9ab59cf3a3d9d9e5a19cbcf53a731bb6a569cb245"},"motivation":"Large language models (LLMs) have recently advanced text-driven 3D generation, yet Text-to-CAD remains far from supporting industrial product design.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28579","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MUSE Benchmark Team","organizationType":"academic-lab","sourceUrl":"https://dong7313.github.io/muse-benchmark/","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_musebench_7c930c72","familyId":"bmf_a54402b9054c","name":"MuseBench","oneLine":"MuseBench evaluates multimodal large language models on intent-level understanding of audiovisual arts, covering cinematic arts, static visual arts, stage performing arts, and game arts with 4,016 questions in single- and multi-select formats.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.30026","pdf":"https://arxiv.org/pdf/2606.30026","project":"https://musebench.github.io","code":"https://github.com/musebench/musebench-code","data":null,"hfPaper":"https://huggingface.co/papers/2606.30026"},"evidence":{"snippet":"To address this gap, we introduce Musebench, a comprehensive benchmark designed to evaluate MLLMs on nuanced artistic understanding.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":6,"hfDailySubmittedAt":"2026-07-08T00:00:00.000Z","githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.30026"},"ranking":{"90d":{"score":34,"rank":208,"coverage":0.7,"confidence":"Medium"}},"description":"MuseBench evaluates multimodal large language models on intent-level understanding of audiovisual arts, covering cinematic arts, static visual arts, stage performing arts, and game arts with 4,016 questions in single- and multi-select formats.","whyItMatters":"Existing video and multimodal benchmarks mostly test perceptual recognition, leaving artistic intent and creative reasoning unevaluated. MuseBench measures a distinct capability gap, providing a reference for progress in creative-domain expertise.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"02efd189ac6f4f0d72cba16da7eff21a3083aee28aff4b4f3be33048ddf67c96"},"motivation":"Audiovisual arts encompass diverse creative disciplines, including cinema, visual arts, stage performance, and game design, where artistic meaning arises from deliberate combinations of visual, auditory, and narrative elements (e.g., fear amplified through claustrophobic framing, or grief conveyed through silence and lingering close-ups).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.30026","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MuseBench project","organizationType":"academic-lab","sourceUrl":"https://github.com/musebench/musebench-code","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_54edb2991e31006c","familyId":"catalog_family_54edb2991e31006c","name":"MusicCaps","oneLine":"MusicCaps is a dataset composed of 5,521 music examples, each labeled with an English aspect list and a free text caption written by musicians. The dataset contains 10-second music clips from AudioSet paired with rich textual descriptions that capture sonic qualities and musical elements like genre, mood, tempo, instrumentation, and rhythm. Created to support research in music-text understanding and generation tasks.","description":"MusicCaps is a dataset composed of 5,521 music examples, each labeled with an English aspect list and a free text caption written by musicians. The dataset contains 10-second music clips from AudioSet paired with rich textual descriptions that capture sonic qualities and musical elements like genre, mood, tempo, instrumentation, and rhythm. Created to support research in music-text understanding and generation tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/musiccaps","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_54edb2991e31006c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/musiccaps"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"musiccaps","url":"https://llm-stats.com/benchmarks/musiccaps","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","audio"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_musp-bench_8c379446","familyId":"bmf_45c04f063445","name":"MuSP-Bench","oneLine":"490-question benchmark for musical score understanding, performance listening, and combined score-performance reasoning, with multiple input modalities and accepted ground-truth answers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://huggingface.co/datasets/bryel-labs/MuSP-Bench","pdf":null,"project":null,"code":null,"data":"https://huggingface.co/datasets/bryel-labs/MuSP-Bench","hfPaper":null},"evidence":{"snippet":"MuSP-Bench MuSP-Bench is a 490-question benchmark for musical score understanding, performance listening, and combined score-performance reasoning.","reasonCodes":["discovered via huggingface","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":208,"hfDatasetLikes":1},"source":{"type":"huggingface","id":"huggingface:bryel-labs/musp-bench"},"ranking":{"30d":{"score":51,"rank":24,"coverage":0.15,"confidence":"Low","datasetDownloadRank":13,"datasetRankPopulation":30},"90d":{"score":48,"rank":79,"coverage":0.3,"confidence":"Low","datasetDownloadRank":37,"datasetRankPopulation":66}},"description":"490-question benchmark for musical score understanding, performance listening, and combined score-performance reasoning, with multiple input modalities and accepted ground-truth answers.","whyItMatters":"Provides a structured way to compare multimodal models on musical score and performance understanding, enabling systematic evaluation across different input representations and task horizons.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"af2ad0ef979628a45c2421f1346b1244148acffcea5ee6facce4d54b5ee0665a"},"motivation":"MuSP-Bench MuSP-Bench is a 490-question benchmark for musical score understanding, performance listening, and combined score-performance reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://huggingface.co/datasets/bryel-labs/MuSP-Bench","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"attentionForecast":{"score":45,"confidence":"Medium","horizon":"7d","reason":"The dataset is publicly accessible with detailed documentation and baseline results, but the domain is niche and the publisher is not widely known.","additionalProperties":false},"evaluationMode":"public_reusable","publishers":[{"name":"bryel-labs","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/datasets/bryel-labs/MuSP-Bench","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_c6bc068e5f9ecdf8","familyId":"catalog_family_c6bc068e5f9ecdf8","name":"MuSR","oneLine":"MuSR (Multistep Soft Reasoning) is a benchmark for evaluating language models on multistep soft reasoning tasks specified in natural language narratives. Created through a neurosymbolic synthetic-to-natural generation algorithm, it generates complex reasoning scenarios like murder mysteries roughly 1000 words in length that challenge current LLMs including GPT-4. The benchmark tests chain-of-thought reasoning capabilities across domains involving commonsense reasoning about physical and social situations.","description":"MuSR (Multistep Soft Reasoning) is a benchmark for evaluating language models on multistep soft reasoning tasks specified in natural language narratives. Created through a neurosymbolic synthetic-to-natural generation algorithm, it generates complex reasoning scenarios like murder mysteries roughly 1000 words in length that challenge current LLMs including GPT-4. The benchmark tests chain-of-thought reasoning capabilities across domains involving commonsense reasoning about physical and social situations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2310.16049","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c6bc068e5f9ecdf8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/musr"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/musr"}],"catalogSources":[{"catalog":"benchlm","sourceId":"musr","url":"https://benchlm.ai/benchmarks/musr","paperUrl":"https://arxiv.org/abs/2310.16049","year":"2023","fullName":"Testing the Limits of Chain-of-thought with Multistep Soft Reasoning","format":"Narrative-based reasoning","tasks":"Multi-step reasoning","successorKey":null},{"catalog":"llm-stats","sourceId":"musr","url":"https://llm-stats.com/benchmarks/musr","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_mustbench_8ac41de6","familyId":"bmf_246fae86adcc","name":"MusTBENCH","oneLine":"MusTBENCH evaluates temporal grounding in Large Audio-Language Models through five temporally grounded question-answering tasks, validated by music experts, to test alignment with audio regions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29300","pdf":"https://arxiv.org/pdf/2605.29300","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29300"},"evidence":{"snippet":"To address this gap, we introduce MusTBENCH, a music-expert-validated benchmark designed to evaluate temporal grounding in LALMs through five temporally grounded question-answering tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29300"},"ranking":{},"description":"MusTBENCH evaluates temporal grounding in Large Audio-Language Models through five temporally grounded question-answering tasks, validated by music experts, to test alignment with audio regions.","whyItMatters":"Temporal grounding is critical for music understanding, where events are localized. MusTBENCH establishes this as a missing capability and offers a challenging benchmark for improvement.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"53e6a10d7325b477c2f3a6e990e56a1628109bac0a3199f6131f7c9d651ec7b7"},"motivation":"Recent Large Audio-Language Models (LALMs) have demonstrated promising abilities in understanding musical content.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29300","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_mv-bench_0da16fba","familyId":"bmf_eebf6373b8cd","name":"MV-Bench","oneLine":"MV-Bench evaluates multimodal language models on coordinating multi-view interface construction using Tableau workbooks, converting specifications into executable web interfaces.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19910","pdf":"https://arxiv.org/pdf/2607.19910","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19910"},"evidence":{"snippet":"We introduce MV-Bench, a benchmark for evaluating MLLMs on coordinated multi-view interface construction.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19910"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"MV-Bench evaluates multimodal language models on coordinating multi-view interface construction using Tableau workbooks, converting specifications into executable web interfaces.","whyItMatters":"Multimodal models increasingly generate code from visual designs, but existing evaluations focus on single-chart generation. A dedicated benchmark assesses coordination and data semantics in multi-view interfaces.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0179b686f733b11b58d53f09353812ab823283c892bce208254d2819f6c7f385"},"motivation":"Multimodal large language models (MLLMs) are increasingly expected to automate visualization development by generating code directly from visual designs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19910","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"mvbench","url":"https://llm-stats.com/benchmarks/mvbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","spatial reasoning","video","vision"],"catalogModelCount":18,"catalogStarCount":0},{"id":"bm_mv2-multi-view-multi-vehicle-driving-datas_f70ec8fb","familyId":"bmf_7d6c3b465beb","name":"MV2","oneLine":"Evaluates novel view synthesis under large viewpoint changes using synchronized captures from car, scooter, and drone across 50 scenes.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.12442","pdf":"https://arxiv.org/pdf/2608.12442","project":"https://mv2-dataset.github.io/","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce the Multi-View Multi-Vehicle (MV2) dataset and benchmark for evaluating NVS models under large viewpoint changes in dynamic urban scenes.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12442"},"ranking":{"30d":{"score":39,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates novel view synthesis under large viewpoint changes using synchronized captures from car, scooter, and drone across 50 scenes.","whyItMatters":"It challenges NVS models on dynamic urban scenes with multi-trajectory disparities beyond existing datasets.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"a3f6fc97af8ff7ecb450e9c1bff6a4c42a5ec3163b79e131c33d43c8a6a9fe24"},"motivation":"Differentiable rendering has advanced novel view synthesis (NVS), yet applying it to real-world driving remains difficult due to sparse capture viewpoints, dynamic objects, and limited multi-trajectory data.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"MV2 is a named dataset and benchmark with project page and protocol, supporting reproducible evaluation.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce the Multi-View Multi-Vehicle (MV2) dataset and benchmark"},"publication":{"status":"acceptance_claimed","venue":"paper","evidence":"18 pages, 7 figures, ECCV accepted paper","evidenceUrl":"https://arxiv.org/abs/2608.12442","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-19T11:05:13.395391Z"},"venueAttempts":[{"venueName":"paper","reviewStatus":"accepted","decisionRaw":"18 pages, 7 figures, ECCV accepted paper","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.12442","observedAt":"2026-08-19T11:05:13.395391Z","rawValue":"18 pages, 7 figures, ECCV accepted paper","level":"author-claim"}]}],"attentionForecast":{"score":63,"confidence":"Medium","horizon":"7d","reason":"MV2 focuses on a specialized but active NVS area with a public dataset, likely drawing interest from computer vision researchers."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_24f36aa0e4b2ebf0","familyId":"catalog_family_24f36aa0e4b2ebf0","name":"NanoBEIR Multilingual","oneLine":"A display-only multilingual retrieval benchmark reported by Liquid AI for LFM2.5 retriever models, using NDCG@10 across 11 languages.","description":"A display-only multilingual retrieval benchmark reported by Liquid AI for LFM2.5 retriever models, using NDCG@10 across 11 languages.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multilingual"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.liquid.ai/blog/lfm2-5-retrievers","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_24f36aa0e4b2ebf0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/nanobeirmultilingual"}],"catalogSources":[{"catalog":"benchlm","sourceId":"nanoBeirMultilingual","url":"https://benchlm.ai/benchmarks/nanobeirmultilingual","paperUrl":"https://www.liquid.ai/blog/lfm2-5-retrievers","year":"2026","fullName":"NanoBEIR Multilingual Extended","format":"NDCG@10 average","tasks":"Multilingual document retrieval","successorKey":null}],"catalogCategories":["multilingual"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_bf694ecef574797c","familyId":"catalog_family_bf694ecef574797c","name":"NanoGPT","oneLine":"NanoGPT is an OpenAI AI-self-improvement evaluation that measures whether models can optimize training recipes for small GPT-style models, part of the suite tracking progress toward accelerating internal research.","description":"NanoGPT is an OpenAI AI-self-improvement evaluation that measures whether models can optimize training recipes for small GPT-style models, part of the suite tracking progress toward accelerating internal research.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code","Systems"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/nanogpt","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bf694ecef574797c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/nanogpt"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"nanogpt","url":"https://llm-stats.com/benchmarks/nanogpt","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code","systems"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_narrativeworldbench_d47453bd","familyId":"bmf_402c51ce9e0f","name":"NarrativeWorldBench","oneLine":"NarrativeWorldBench evaluates long-horizon narrative generation in audio drama across nine structural metrics and four Indic languages, across horizons from 10 to 200 episodes. Scoring is based on plot-beat F1 and other metrics.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.17391","pdf":"https://arxiv.org/pdf/2606.17391","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.17391"},"evidence":{"snippet":"We introduce NarrativeWorldBench, an open benchmark of nine narrative-structure metrics evaluated across horizons h in {10, 20, 50, 100, 200}, with cross-lingual evaluation across four Indic languages (Hindi, Tamil, Telugu, Marathi).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.17391"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"NarrativeWorldBench evaluates long-horizon narrative generation in audio drama across nine structural metrics and four Indic languages, across horizons from 10 to 200 episodes. Scoring is based on plot-beat F1 and other metrics.","whyItMatters":"Current LLMs degrade on long-horizon narrative coherence. This benchmark provides metrics for evaluating long-form structured content generation and cross-lingual capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"6996f6ece7ba8c94674684b0437947afc552b16fa5701464ca2ce45ce5b07017"},"motivation":"Long-form serialized audio drama, with arcs that run for 200 to 800 episodes, is a major creative medium and a setting where frontier large language models (LLMs) fail.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026 Workshops on High-dimensional Learning Dynamics (HiLD) and Culture x AI","evidence":"10 pages. Accepted to the ICML 2026 Workshops on High-dimensional Learning Dynamics (HiLD) and Culture x AI","evidenceUrl":"https://arxiv.org/abs/2606.17391","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026 Workshops on High-dimensional Learning Dynamics (HiLD) and Culture x AI","reviewStatus":"accepted","decisionRaw":"10 pages. Accepted to the ICML 2026 Workshops on High-dimensional Learning Dynamics (HiLD) and Culture x AI","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.17391","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"10 pages. Accepted to the ICML 2026 Workshops on High-dimensional Learning Dynamics (HiLD) and Culture x AI","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_naru_dcb9acb4","familyId":"bmf_62e65dd5c196","name":"NARU","oneLine":"NARU evaluates multimodal models on Japanese long-form video understanding across narrative evolution (character evolution, sequential flow, plot progression, thematic development) and cultural nuance (aizuchi, reading the air, subtext, cultural context, sentiment) via multiple-choice QA.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.13210","pdf":"https://arxiv.org/pdf/2608.13210","project":"https://ma-labo.github.io/naru/","code":"https://github.com/infinimind-inc/naru_benchmark","data":null,"hfPaper":"https://huggingface.co/papers/2608.13210"},"evidence":{"snippet":"To address this gap, we introduce NARU, a benchmark designed to evaluate Narrative evolution and Reasoning on cultural Understanding in Japanese long-form video.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":9,"hfDailySubmittedAt":"2026-08-21T00:00:00.000Z","githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13210"},"ranking":{"30d":{"score":31,"rank":79,"coverage":0.85,"confidence":"High"},"90d":{"score":32,"rank":223,"coverage":0.7,"confidence":"Medium"}},"description":"NARU evaluates multimodal models on Japanese long-form video understanding across narrative evolution (character evolution, sequential flow, plot progression, thematic development) and cultural nuance (aizuchi, reading the air, subtext, cultural context, sentiment) via multiple-choice QA.","whyItMatters":"NARU fills a gap in video QA benchmarks by jointly testing long-range narrative tracking and culturally grounded reasoning, which are absent in existing short-horizon or English-centric benchmarks. It provides a rigorous, reproducible protocol for measuring model capabilities in high-context media, supporting model development and product evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-22T11:34:55.479945Z","inputHash":"62498a267d98c1ff82a21e867814665020b942e0ce2b724263a3fff78ebe39be"},"motivation":"Long-form video understanding encompasses tasks that go beyond retrieving isolated events, including tracking an evolving narrative and interpreting social meaning that may remain implicit.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13210","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Infinimind Inc.","organizationType":"company-research-lab","sourceUrl":"https://github.com/infinimind-inc/naru_benchmark","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_db81694b1eca554d","familyId":"catalog_family_db81694b1eca554d","name":"Natural Questions","oneLine":"Natural Questions is a question answering dataset featuring real anonymized queries issued to Google search engine. It contains 307,373 training examples where annotators provide long answers (passages) and short answers (entities) from Wikipedia pages, or mark them as unanswerable.","description":"Natural Questions is a question answering dataset featuring real anonymized queries issued to Google search engine. It contains 307,373 training examples where annotators provide long answers (passages) and short answers (entities) from Wikipedia pages, or mark them as unanswerable.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Search","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/natural-questions","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_db81694b1eca554d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/natural-questions"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"natural-questions","url":"https://llm-stats.com/benchmarks/natural-questions","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","search","general"],"catalogModelCount":7,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_c12e1090234d9345","familyId":"catalog_family_c12e1090234d9345","name":"Natural2Code","oneLine":"NaturalCodeBench (NCB) is a challenging code benchmark designed to mirror the complexity and variety of real-world coding tasks. It comprises 402 high-quality problems in Python and Java, selected from natural user queries from online coding services, covering 6 different domains.","description":"NaturalCodeBench (NCB) is a challenging code benchmark designed to mirror the complexity and variety of real-world coding tasks. It comprises 402 high-quality problems in Python and Java, selected from natural user queries from online coding services, covering 6 different domains.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/natural2code","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c12e1090234d9345"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/natural2code"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"natural2code","url":"https://llm-stats.com/benchmarks/natural2code","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_naturebench_ffd14b37","familyId":"bmf_d35b766e0cf9","name":"NatureBench","oneLine":"NatureBench evaluates AI coding agents on 90 tasks distilled from Nature-family publications across 6 scientific domains, scoring against each paper's reported state of the art.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24530","pdf":"https://arxiv.org/pdf/2606.24530","project":null,"code":"https://github.com/FrontisAI/NatureBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.24530"},"evidence":{"snippet":"We introduce NatureBench, a cross-discipline benchmark of 90 tasks distilled from peer-reviewed Nature-family publications, designed to evaluate whether AI coding agents can move beyond reproduction toward discovery on real scientific problems.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":66,"hfDailySubmittedAt":"2026-06-24T00:00:00.000Z","githubStars":106,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24530"},"ranking":{"90d":{"score":61,"rank":17,"coverage":0.7,"confidence":"Medium"}},"description":"NatureBench evaluates AI coding agents on 90 tasks distilled from Nature-family publications across 6 scientific domains, scoring against each paper's reported state of the art.","whyItMatters":"Provides a standardized environment and public leaderboard to measure whether coding agents can achieve discovery-level performance on real scientific problems, addressing environment-fragmentation issues in prior benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"17b5881db23bbb713bc8a2955bc962a45b051e371c9b4ea34540417320ecdf94"},"motivation":"We introduce NatureBench, a cross-discipline benchmark of 90 tasks distilled from peer-reviewed Nature-family publications, designed to evaluate whether AI coding agents can move beyond reproduction toward discovery on real scientific problems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24530","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FrontisAI","organizationType":"company-research-lab","sourceUrl":"https://github.com/FrontisAI/NatureBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_navverse_d8d5fb61","familyId":"bmf_d28dc1824ef0","name":"NavVerse","oneLine":"NavVerse evaluates indoor-to-outdoor embodied navigation in continuous robot simulation, with 10,000 episodes across object, vision-language, and place navigation tasks.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19695","pdf":"https://arxiv.org/pdf/2607.19695","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19695"},"evidence":{"snippet":"We introduce NavVerse, a physics-enabled benchmark for indoor-to-outdoor embodied navigation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19695"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"NavVerse evaluates indoor-to-outdoor embodied navigation in continuous robot simulation, with 10,000 episodes across object, vision-language, and place navigation tasks.","whyItMatters":"Existing benchmarks evaluate indoor and outdoor navigation separately, missing cross-context challenges. A dedicated benchmark supports progress in integrated navigation scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"025fdf01ff32bd5cb253b2e1d67639c246d4b6e0c8db5028d745afe9d5b75636"},"motivation":"Robots deployed in delivery, campus, and emergency-response settings often need to navigate from buildings to streets within a single continuous episode.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19695","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_nba-streaming_a4cd00b8","familyId":"bmf_c0442d23bb74","name":"NBA_Streaming","oneLine":"NBA_Streaming is a benchmark for online fine-grained basketball commentary generation, containing 307.5 hours of broadcasts and approximately 35K temporally aligned events with annotations of event boundaries, player identities, fine-grained actions, event chains, and natural-language commentary.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09200","pdf":"https://arxiv.org/pdf/2608.09200","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09200"},"evidence":{"snippet":"To address these limitations, we introduce NBA_Streaming, a large-scale benchmark for online fine-grained basketball commentary generation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09200"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"NBA_Streaming is a benchmark for online fine-grained basketball commentary generation, containing 307.5 hours of broadcasts and approximately 35K temporally aligned events with annotations of event boundaries, player identities, fine-grained actions, event chains, and natural-language commentary.","whyItMatters":"It addresses the evaluation gap in streaming sports video understanding and generation, enabling unified assessment of event localization, response reliability, factual grounding, and commentary quality under causal constraints.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0c7eb2c6d9c31457a37d3d9f57524a8312faa1faa4466562e40fdcfd95de5d42"},"motivation":"Live basketball commentary generation requires determining when an event is sufficiently observable and describing it before subsequent events unfold.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09200","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_ncp-bench_83c91412","familyId":"bmf_b98627ff590e","name":"NCP-Bench","oneLine":"NCP-Bench evaluates narrative commitment preservation in interactive narratives. It provides 100 movie-synopsis-based environments with structured narrative specifications and an automatic evaluator that checks fact, commitment, and trajectory consistency across narrator responses to player actions.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-08","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.08160","pdf":"https://arxiv.org/pdf/2608.08160","project":null,"code":"https://github.com/yingpengma/NCP-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.08160"},"evidence":{"snippet":"We introduce NCP-Bench, a benchmark of 100 narrative environments derived from movie synopses.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":29,"hfDailySubmittedAt":"2026-08-13T00:00:00.000Z","githubStars":32,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08160"},"ranking":{"30d":{"score":54,"rank":14,"coverage":0.85,"confidence":"High"},"90d":{"score":50,"rank":64,"coverage":0.7,"confidence":"Medium"}},"description":"NCP-Bench evaluates narrative commitment preservation in interactive narratives. It provides 100 movie-synopsis-based environments with structured narrative specifications and an automatic evaluator that checks fact, commitment, and trajectory consistency across narrator responses to player actions.","whyItMatters":"The benchmark fills a gap in evaluating long-horizon logical consistency of LLM-driven interactive narrators, offering a repeatable protocol to compare models on narrative integrity under adversarial user interventions. This supports practical selection of models for interactive storytelling applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"61b0d5cc322af2e5dcd34aabfcf5d9257813a42e0dbee2d2254ad8a895921b72"},"motivation":"The rapid advancement of Large Language Models (LLMs) is revolutionizing AI for Games by enabling open-ended and fluid interactive storytelling.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026","evidence":"Accepted by ICML 2026","evidenceUrl":"https://arxiv.org/abs/2608.08160","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026","reviewStatus":"accepted","decisionRaw":"Accepted by ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.08160","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by ICML 2026","level":"author-claim"}]}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_needl-bench_c98901f1","familyId":"bmf_825f0496303e","name":"NEEDL-Bench","oneLine":"NEEDL-Bench is a microscopy detection benchmark for Swiss Needle Cast and stomata detection, with 3250 annotated images from 1082 Douglas-fir needles, annotated for keypoint and bounding-box detectors, and two evaluation splits.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.12076","pdf":"https://arxiv.org/pdf/2607.12076","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.12076"},"evidence":{"snippet":"We present NEEDL-Bench, a microscopy detection benchmark for Swiss Needle Cast (SNC), a fungal disease of Douglas-fir trees.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.12076"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"NEEDL-Bench is a microscopy detection benchmark for Swiss Needle Cast and stomata detection, with 3250 annotated images from 1082 Douglas-fir needles, annotated for keypoint and bounding-box detectors, and two evaluation splits.","whyItMatters":"There is no existing dataset for automatic detection of these structures, despite the importance of Douglas-fir. This benchmark provides a standardized evaluation for keypoint and object detection methods, highlighting challenges like small objects and occlusion.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"61ba737d930d2e4dd0e4c14423206d07714785c361a5776a829267275e8238d4"},"motivation":"We present NEEDL-Bench, a microscopy detection benchmark for Swiss Needle Cast (SNC), a fungal disease of Douglas-fir trees.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.12076","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_netconfarena_7ff1c051","familyId":"bmf_78cd2ef5a953","name":"NetConfArena","oneLine":"Evaluates LLM agents in closed-loop network configuration across 480 task instances from 96 protocol-focused templates, scoring based on hidden executable test cases.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.NI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.23179","pdf":"https://arxiv.org/pdf/2608.23179","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present NetConfArena, an executable benchmark for evaluating LLM agents in closed-loop network configuration.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23179"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLM agents in closed-loop network configuration across 480 task instances from 96 protocol-focused templates, scoring based on hidden executable test cases.","whyItMatters":"Provides a realistic, executable benchmark for network configuration agents, revealing failure patterns and guiding improvements in agent reliability and planning.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"27780fd4b840aab3604da4511364f72da1c3c435fc5d1b74b2f4b48aa9a504af"},"motivation":"Large language model (LLM) agents are increasingly attractive for automating network configuration, yet their reliability and failure patterns are poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The paper presents a named, executable benchmark with a clear evaluation protocol and public availability, indicating an ongoing public benchmark for model comparison.","canonicalNameSource":"abstract","canonicalNameEvidence":"We present NetConfArena, an executable benchmark for evaluating LLM agents in closed-loop network configuration."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.23179","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":70,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses a critical infrastructure domain with an interactive evaluation setting, likely to attract interest from both LLM agent and networking communities."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_netinjectbench_1ee2f598","familyId":"bmf_e8dac394667e","name":"NetInjectBench","oneLine":"NetInjectBench evaluates LLM agents for network operations under indirect prompt injection. It comprises 130 scenarios: 40 benign, 40 weak-attack, 40 strong-attack, and 10 approved high-impact changes, with separated untrusted artifact text, trusted policy metadata, and evaluation labels. Scoring measures unsafe tool-action rate, usefulness, and overblocking across models and defenses.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.10490","pdf":"https://arxiv.org/pdf/2607.10490","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.10490"},"evidence":{"snippet":"We present NetInjectBench, a 130-scenario benchmark that separates untrusted artifact text, trusted policy metadata, and evaluation labels for network-operation tool use.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.10490"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"NetInjectBench evaluates LLM agents for network operations under indirect prompt injection. It comprises 130 scenarios: 40 benign, 40 weak-attack, 40 strong-attack, and 10 approved high-impact changes, with separated untrusted artifact text, trusted policy metadata, and evaluation labels. Scoring measures unsafe tool-action rate, usefulness, and overblocking across models and defenses.","whyItMatters":"Network operations increasingly rely on tool-using LLM agents, which are vulnerable to indirect prompt injections via untrusted artifacts. This benchmark quantifies safety and usefulness trade-offs in a realistic network-domain environment, providing a concrete means to assess defense effectiveness and authorization boundaries.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3485d8e24f03ea71ead158ee6e58d3d801744dc00caccf50b513fe55e82b8ace"},"motivation":"Tool-using large language model (LLM) agents are attractive for network operations, but tickets, alerts, logs, runbooks, and ChatOps messages can carry indirect prompt injections.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.10490","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_netlistbench_58d2657e","familyId":"bmf_3c373592114f","name":"NetlistBench","oneLine":"NetlistBench evaluates LLM reliability in recognizing and manipulating SPICE netlists, covering parameter and connectivity recognition, edits, hierarchical operations, equivalence judgment, and compound editing, with deterministic structure-aware scoring.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12197","pdf":"https://arxiv.org/pdf/2608.12197","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12197"},"evidence":{"snippet":"We present \\textbf{NetlistBench}, a structure-verified benchmark for SPICE netlist recognition and manipulation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12197"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"NetlistBench evaluates LLM reliability in recognizing and manipulating SPICE netlists, covering parameter and connectivity recognition, edits, hierarchical operations, equivalence judgment, and compound editing, with deterministic structure-aware scoring.","whyItMatters":"It addresses the gap in assessing LLM reliability for simulator-facing netlist tasks, distinct from high-level design reasoning, and provides a structured evaluation to guide trustworthy circuit design automation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"690dac1af652a91d5cdcc0ce216c9c61fd24dddc9bbcf0cfeadbcdfa0163eadf"},"motivation":"Large Language Models (LLMs) are increasingly used in circuit design workflows, yet their reliability on simulator-facing SPICE netlist recognition and manipulation remains poorly understood and is rarely separated from high-level design reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"MLCAD 2026","evidence":"accepted by MLCAD 2026","evidenceUrl":"https://arxiv.org/abs/2608.12197","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"MLCAD 2026","reviewStatus":"accepted","decisionRaw":"accepted by MLCAD 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.12197","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"accepted by MLCAD 2026","level":"author-claim"}]}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_neurai-vn-benchmark_10fca573","familyId":"bmf_5f53938b8ce8","name":"Neurai-VN Benchmark","oneLine":"Evaluates machine learning models on the Neurai-VN dataset for mental health classification. Four binary tasks (healthy control vs. depression, anxiety, clinical, and depression vs. anxiety) are defined using subject-wise cross-validation and standardized feature groups. Baseline models include linear, tree-based, and neural networks, with mean F1 scores reported.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25232","pdf":"https://arxiv.org/pdf/2607.25232","project":null,"code":"https://github.com/neurai-vn/Neurai-VN-benchmark","data":null,"hfPaper":"https://huggingface.co/papers/2607.25232"},"evidence":{"snippet":"In this work, we introduce a reproducible machine learning benchmark using the Neurai-VN dataset, a multimodal digital phenotyping dataset collected from 100 Vietnamese adults over two weeks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25232"},"ranking":{"90d":{"score":31,"rank":234,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates machine learning models on the Neurai-VN dataset for mental health classification. Four binary tasks (healthy control vs. depression, anxiety, clinical, and depression vs. anxiety) are defined using subject-wise cross-validation and standardized feature groups. Baseline models include linear, tree-based, and neural networks, with mean F1 scores reported.","whyItMatters":"Provides a standardized evaluation protocol for multimodal digital phenotyping in mental health, addressing inconsistencies in preprocessing and evaluation across datasets. Offers comparable baselines for future research.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f6fb97748d48ddc3666de11ab838b1f6252ec4d1d47ca8b2e9830cacdda2d1a6"},"motivation":"Digital phenotyping (DP) using smartphones and wearable devices has emerged as a promising approach for assessing mental health, particularly depression and anxiety.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25232","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Neurai-VN","organizationType":"community","sourceUrl":"https://github.com/neurai-vn/Neurai-VN-benchmark","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_d59cd44e49897416","familyId":"catalog_family_d59cd44e49897416","name":"Next.js Evals","oneLine":"A Vercel benchmark for AI coding agents on Next.js code generation and migration tasks, reporting success rate, average execution time, and an AGENTS.md documentation-assisted split.","description":"A Vercel benchmark for AI coding agents on Next.js code generation and migration tasks, reporting success rate, average execution time, and an AGENTS.md documentation-assisted split.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://nextjs.org/evals","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d59cd44e49897416"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/nextjsevals"}],"catalogSources":[{"catalog":"benchlm","sourceId":"nextjsEvals","url":"https://benchlm.ai/benchmarks/nextjsevals","paperUrl":"https://nextjs.org/evals","year":"2026","fullName":"AI Agent Evaluations for Next.js","format":"Agent task completion with withheld Vitest assertions","tasks":"24 Next.js code generation and migration tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_nextmotionqa_b6672a26","familyId":"bmf_cfb80271e1dd","name":"NextMotionQA","oneLine":"NextMotionQA is a benchmark for human motion understanding with VLMs, featuring three tasks: multiple-choice QA, video captioning, and fine-grained error correction. It includes expert-verified annotations across three semantic axes and three complexity levels.","area":"Multimodal","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04773","pdf":"https://arxiv.org/pdf/2606.04773","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04773"},"evidence":{"snippet":"To bridge this gap, we introduce NextMotionQA, a comprehensive benchmark that leverages vision-language models (VLMs) for semi-automated, expert-verified dataset.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04773"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"NextMotionQA is a benchmark for human motion understanding with VLMs, featuring three tasks: multiple-choice QA, video captioning, and fine-grained error correction. It includes expert-verified annotations across three semantic axes and three complexity levels.","whyItMatters":"Existing motion benchmarks have coarse granularity and ambiguity. NextMotionQA provides structured tasks across complexity levels, enabling diagnosis of VLM capability gaps in motion understanding and judging.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e1f87425cff169224320dfaab39b5fd3689c3313d1d31a9b59365f5ca7adce53"},"motivation":"Reliable evaluation of human motion understanding is fundamental to advancing embodied AI, robotics, and animation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04773","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_f5cfcb570b7edac2","familyId":"catalog_family_f5cfcb570b7edac2","name":"Nexus","oneLine":"NexusRaven benchmark for evaluating function calling capabilities of large language models in zero-shot scenarios across cybersecurity tools and API interactions","description":"NexusRaven benchmark for evaluating function calling capabilities of large language models in zero-shot scenarios across cybersecurity tools and API interactions","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["General","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/nexus","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f5cfcb570b7edac2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/nexus"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"nexus","url":"https://llm-stats.com/benchmarks/nexus","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["general","tool calling"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_1affa89e646bc694","familyId":"catalog_family_1affa89e646bc694","name":"NIH/Multi-needle","oneLine":"Multi-needle in a haystack benchmark for evaluating long-context comprehension capabilities of language models by testing retrieval of multiple target pieces of information from extended documents","description":"Multi-needle in a haystack benchmark for evaluating long-context comprehension capabilities of language models by testing retrieval of multiple target pieces of information from extended documents","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/nih-multi-needle","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1affa89e646bc694"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/nih-multi-needle"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"nih-multi-needle","url":"https://llm-stats.com/benchmarks/nih-multi-needle","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_nl-pddl-bench_3b0f6c71","familyId":"bmf_57cc06c1d191","name":"NL-PDDL-Bench","oneLine":"NL-PDDL-Bench evaluates natural-language-to-PDDL specification generation. It consists of multi-domain instances from IPC domains with planner-verified executability, difficulty scaled by object count, and a suite for parseability, solvability, and plan-level consistency.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.29700","pdf":"https://arxiv.org/pdf/2606.29700","project":null,"code":"https://github.com/ibasicplan/NL-PDDL-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.29700"},"evidence":{"snippet":"We present NL-PDDL-Bench, a multi-domain benchmark for natural-language-to-PDDL specification construction with planner-verified executability and controlled difficulty scaling by object count.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29700"},"ranking":{"90d":{"score":28,"rank":278,"coverage":0.55,"confidence":"Low"}},"description":"NL-PDDL-Bench evaluates natural-language-to-PDDL specification generation. It consists of multi-domain instances from IPC domains with planner-verified executability, difficulty scaled by object count, and a suite for parseability, solvability, and plan-level consistency.","whyItMatters":"This benchmark addresses the lack of standardized evaluation for LLM-generated planning specifications, with an emphasis on executability and verifiability. It provides a reproducible basis for assessing model reliability in safety-sensitive planning applications, where incorrect formalization can lead to unsafe outcomes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4f45b825d0fb9bfbde2b16b0422ddd99354885540e2916e519ecc955809c66c9"},"motivation":"Planning often requires symbolic specifications that are both executable and verifiable.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.29700","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ibasicplan","organizationType":"academic-lab","sourceUrl":"https://github.com/ibasicplan/NL-PDDL-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_1d870239ba679184","familyId":"catalog_family_1d870239ba679184","name":"NL2Repo","oneLine":"NL2Repo evaluates long-horizon coding capabilities including repository-level understanding, where models must generate or modify code across entire repositories from natural language specifications.","description":"NL2Repo evaluates long-horizon coding capabilities including repository-level understanding, where models must generate or modify code across entire repositories from natural language specifications.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.minimax.io/news/minimax-m27-en","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1d870239ba679184"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/nl2repo"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/nl2repo"}],"catalogSources":[{"catalog":"benchlm","sourceId":"nl2Repo","url":"https://benchlm.ai/benchmarks/nl2repo","paperUrl":"https://www.minimax.io/news/minimax-m27-en","year":"2026","fullName":"NL2Repo","format":"Repository understanding benchmark","tasks":"Natural language to repository tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"nl2repo","url":"https://llm-stats.com/benchmarks/nl2repo","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","agents","code"],"catalogModelCount":20,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_nl2scratch_fe2ce42e","familyId":"bmf_3590e32806d4","name":"NL2Scratch","oneLine":"NL2Scratch is an executable benchmark for natural-language-to-Scratch generation, consisting of 311,648 parser-valid NL-program pairs extracted from real Scratch projects. It includes a semantically validated pool of 23,594 examples and an 800-example diagnostic set. Evaluation uses Semantic Alignment Consistency (SAC), an interpretable slot-level metric for measuring semantic agreement.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Code generation"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22061","pdf":"https://arxiv.org/pdf/2606.22061","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.22061"},"evidence":{"snippet":"We introduce NL2Scratch, an executable benchmark for natural-language-to-Scratch generation comprising 311,648 parser-valid NL--program pairs, whose program side is extracted from real Scratch projects and paired with semantically aligned NL descriptions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22061"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"NL2Scratch is an executable benchmark for natural-language-to-Scratch generation, consisting of 311,648 parser-valid NL-program pairs extracted from real Scratch projects. It includes a semantically validated pool of 23,594 examples and an 800-example diagnostic set. Evaluation uses Semantic Alignment Consistency (SAC), an interpretable slot-level metric for measuring semantic agreement.","whyItMatters":"Existing NL2Code evaluation focuses on text-based languages, leaving block-based programming unevaluated. NL2Scratch enables assessment of models on event-driven, visually compositional programs, revealing gaps between lexical similarity and semantic alignment that are invisible under token-level metrics.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5f22e85182740424ca0a409b023042bb9a0624de75b20a79512b1d1dfb8e8c59"},"motivation":"Block-based programming environments such as Scratch are widely used in early programming education, yet natural-language-to-code (NL2Code) research has focused primarily on text-based languages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22061","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"NL2Scratch Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.22061","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_nl2shacl-bench_8b61bc16","familyId":"bmf_95864a020f6a","name":"NL2SHACL-Bench","oneLine":"A benchmark suite for translating natural language requirements into SHACL shapes, evaluating semantic equivalence beyond string comparison.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07530","pdf":"https://arxiv.org/pdf/2608.07530","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07530"},"evidence":{"snippet":"To tackle these challenges, we present NL2SHACL-Bench, a benchmark suite for natural language to SHACL translation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07530"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark suite for translating natural language requirements into SHACL shapes, evaluating semantic equivalence beyond string comparison.","whyItMatters":"Provides a meaningful basis for measuring advances in NL2SHACL translation, which is critical for domain experts authoring SHACL constraints.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"18d8452ba30257ec36936d3bd06d6d58874e99afb39f4f4fd8aba91cc13a3431"},"motivation":"SHACL is a core technology for validating the conformance of RDF knowledge graphs (KGs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ISWC 2026","evidence":"Accepted at ISWC 2026; 18 pages, 8 figures, 2 tables","evidenceUrl":"https://arxiv.org/abs/2608.07530","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ISWC 2026","reviewStatus":"accepted","decisionRaw":"Accepted at ISWC 2026; 18 pages, 8 figures, 2 tables","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.07530","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at ISWC 2026; 18 pages, 8 figures, 2 tables","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_57e68ee3e8de805c","familyId":"catalog_family_57e68ee3e8de805c","name":"NMOS","oneLine":"NMOS evaluation benchmark for assessing model performance on specialized tasks","description":"NMOS evaluation benchmark for assessing model performance on specialized tasks","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/nmos","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_57e68ee3e8de805c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/nmos"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"nmos","url":"https://llm-stats.com/benchmarks/nmos","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_00d121bee97ad2d2","familyId":"catalog_family_00d121bee97ad2d2","name":"nolima","oneLine":"Catalog-listed benchmark; original-source verification is pending.","description":"Catalog-listed benchmark; original-source verification is pending.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/community:2256e9c9-b256-4444-b639-7cc3b1855d96","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_00d121bee97ad2d2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/community:2256e9c9-b256-4444-b639-7cc3b1855d96"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"community:2256e9c9-b256-4444-b639-7cc3b1855d96","url":"https://llm-stats.com/benchmarks/community:2256e9c9-b256-4444-b639-7cc3b1855d96","datasetSlug":"nolima","versionCount":2,"subsetCount":10,"rowCount":null,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":["long context"],"catalogModelCount":52,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_4d097a7c6282c603","familyId":"catalog_family_4d097a7c6282c603","name":"NoLiMa 128K","oneLine":"NoLiMa evaluated at a 131072-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.","description":"NoLiMa evaluated at a 131072-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/nolima-128k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4d097a7c6282c603"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/nolima-128k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"nolima-128k","url":"https://llm-stats.com/benchmarks/nolima-128k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_89acbf6e5d1b5b29","familyId":"catalog_family_89acbf6e5d1b5b29","name":"NoLiMa 32K","oneLine":"NoLiMa evaluated at a 32768-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.","description":"NoLiMa evaluated at a 32768-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/nolima-32k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_89acbf6e5d1b5b29"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/nolima-32k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"nolima-32k","url":"https://llm-stats.com/benchmarks/nolima-32k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_7f755de6145eeb6c","familyId":"catalog_family_7f755de6145eeb6c","name":"NoLiMa 64K","oneLine":"NoLiMa evaluated at a 65536-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.","description":"NoLiMa evaluated at a 65536-token context length. Tests latent associative reasoning in long contexts with minimal lexical overlap between questions and needles.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/nolima-64k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7f755de6145eeb6c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/nolima-64k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"nolima-64k","url":"https://llm-stats.com/benchmarks/nolima-64k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_nolli_b5bb5859","familyId":"bmf_9e92e28bd0ca","name":"NOLLI","oneLine":"NOLLI is a procedurally generated English-Korean puzzle benchmark with 15 puzzle types (25 tasks, 7,500 items). Each instance is seed-regenerable, verified to have a unique solution, and scored deterministically. Difficulty is calibrated behaviorally to target accuracy bands.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04397","pdf":"https://arxiv.org/pdf/2608.04397","project":null,"code":"https://github.com/HAE-RAE/NOLLI","data":null,"hfPaper":"https://huggingface.co/papers/2608.04397"},"evidence":{"snippet":"We introduce NOLLI, a procedurally generated English-Korean puzzle benchmark designed to diagnose where Korean performance gaps arise.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":23,"hfDailySubmittedAt":"2026-08-06T00:00:00.000Z","githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04397"},"ranking":{"30d":{"score":39,"rank":50,"coverage":0.85,"confidence":"High"},"90d":{"score":37,"rank":178,"coverage":0.7,"confidence":"Medium"}},"description":"NOLLI is a procedurally generated English-Korean puzzle benchmark with 15 puzzle types (25 tasks, 7,500 items). Each instance is seed-regenerable, verified to have a unique solution, and scored deterministically. Difficulty is calibrated behaviorally to target accuracy bands.","whyItMatters":"NOLLI addresses the lack of controlled cross-lingual benchmarks that separate presentation language from reasoning difficulty, enabling diagnosis of where model performance gaps arise across languages and writing systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d5b263a38d9fcdd605bc1a4663b7748e42e1fd043a37ffb89972162a092a1fd3"},"motivation":"We introduce NOLLI, a procedurally generated English-Korean puzzle benchmark designed to diagnose where Korean performance gaps arise.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04397","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HAE-RAE","organizationType":"academic-lab","sourceUrl":"https://github.com/HAE-RAE/NOLLI","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_nora_3a8754af","familyId":"bmf_b21ec07effd3","name":"NoRA","oneLine":"NoRA is a visual first-person video benchmark requiring models to generate candidate next actions and justify them via fact-reason-action support graphs. It includes 1,420 annotated video clips and uses a grounded reasonableness score combining action alignment, factual grounding, and support binding.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.04806","pdf":"https://arxiv.org/pdf/2606.04806","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04806"},"evidence":{"snippet":"We introduce NoRA, a visual first-person video benchmark that requires models to generate candidate next actions and justify each through an explicit fact-reason-action support graph.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04806"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"NoRA is a visual first-person video benchmark requiring models to generate candidate next actions and justify them via fact-reason-action support graphs. It includes 1,420 annotated video clips and uses a grounded reasonableness score combining action alignment, factual grounding, and support binding.","whyItMatters":"Normative competence in agents requires generating reasonable actions from scratch, grounded in visual facts. NoRA measures this ability and exposes gaps in current VLMs' reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b07665bb95b870918e74a2a082a373be2330e0a822fb2f9fc27183e30d815f4e"},"motivation":"LLMs and agentic systems are increasingly deployed in social environments, making normative competence critical for safe and appropriate behavior.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04806","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_normact_ade65389","familyId":"bmf_a47d8a75095d","name":"NormAct","oneLine":"NormAct evaluates embodied agents on 550 TongSim scenarios where the same goal permits norm-compliant or norm-violating action sequences, testing whether agents infer and apply scene-relevant social norms during ordinary tasks.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.27826","pdf":"https://arxiv.org/pdf/2606.27826","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.27826"},"evidence":{"snippet":"We introduce NormAct, a benchmark of 550 TongSim scenarios in which the same goal permits norm-compliant or norm-violating action sequences.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.27826"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"NormAct evaluates embodied agents on 550 TongSim scenarios where the same goal permits norm-compliant or norm-violating action sequences, testing whether agents infer and apply scene-relevant social norms during ordinary tasks.","whyItMatters":"NormAct addresses the gap between goal achievement and proactive norm compliance, providing a way to assess whether embodied agents respect unstated social norms without prompting, which is critical for real-world deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8e179a3fb04d0764d24772bf6ddb92b34e4c78779e77270cb417dba995d152ab"},"motivation":"Embodied agents driven by multimodal large language models (MLLMs) can often complete everyday tasks from visual observations, but goal achievement does not establish whether they proactively respect unstated social norms.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.27826","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_normbench_a8bf7215","familyId":"bmf_8a7f98f06ad7","name":"NormBench","oneLine":"NormBench evaluates defeasible scope parsing in legal texts using Span-Grounded Deontic Trees, with 2,290 provisions across multiple languages. It focuses on identifying clause overrides and includes whole-tree fidelity metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08932","pdf":"https://arxiv.org/pdf/2606.08932","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.08932"},"evidence":{"snippet":"To diagnose and mitigate SSO, we introduce NormBench, a benchmark of 2,290 provisions spanning Chinese (laws and local policies), English (U.S.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08932"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"NormBench evaluates defeasible scope parsing in legal texts using Span-Grounded Deontic Trees, with 2,290 provisions across multiple languages. It focuses on identifying clause overrides and includes whole-tree fidelity metrics.","whyItMatters":"Silent Scope Omission is a critical failure in rule-following agents. NormBench provides a diagnostic benchmark to identify structural omissions and improve statutory understanding in LLMs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"56dbc27679f4534c96d8a9c38161aeb241021744fead820c3614890513af863d"},"motivation":"Rule-following agents tasked with executing policies and regulations often fail via Silent Scope Omission (SSO): a model applies a general rule but silently drops nested exceptions or counter-exceptions, producing outputs that appear compliant yet break on important edge cases.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08932","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_d98712704c496695","familyId":"catalog_family_d98712704c496695","name":"NOVA-63","oneLine":"NOVA-63 is a multilingual evaluation benchmark covering 63 languages, designed to assess LLM performance across diverse linguistic contexts and tasks.","description":"NOVA-63 is a multilingual evaluation benchmark covering 63 languages, designed to assess LLM performance across diverse linguistic contexts and tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multilingual","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d98712704c496695"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/nova63"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/nova-63"}],"catalogSources":[{"catalog":"benchlm","sourceId":"nova63","url":"https://benchlm.ai/benchmarks/nova63","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"NOVA-63","format":"Cross-lingual benchmark","tasks":"Broad multilingual evaluation","successorKey":null},{"catalog":"llm-stats","sourceId":"nova-63","url":"https://llm-stats.com/benchmarks/nova-63","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multilingual","general"],"catalogModelCount":11,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_novelapibench_5fb13ed7","familyId":"bmf_05bbd7d1ac95","name":"NovelAPIBench","oneLine":"NovelAPIBench is a dynamic benchmark for evaluating LLM tool use with novel APIs, covering knowledge components and diagnostic failure categories.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Tool use","Factuality"],"topics":["Agents"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03657","pdf":"https://arxiv.org/pdf/2606.03657","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03657"},"evidence":{"snippet":"We introduce NovelAPIBench, a fully automated dynamic benchmark that, for any base model and target library, discovers novel APIs, extracts decomposed knowledge bundles, generates executable coding tasks, and assigns failed samples to six diagnostic categories.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03657"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"NovelAPIBench is a dynamic benchmark for evaluating LLM tool use with novel APIs, covering knowledge components and diagnostic failure categories.","whyItMatters":"It addresses the gap in evaluating models' ability to acquire new APIs, providing insights into the complementary roles of retrieval and fine-tuning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"36764ea5906874a108e686ec7760c9b7b3aa6bb5576dfabdf0c32a5e81bec8d1"},"motivation":"Large language models for code generation often need to use APIs that are absent from their pretraining data.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03657","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents","Tool Calling","Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_af1a411ed6a2fdf0","familyId":"catalog_family_af1a411ed6a2fdf0","name":"NQ","oneLine":"Natural Questions (NQ) benchmark containing real user questions issued to Google search with answers found from Wikipedia, designed for training and evaluation of automatic question answering systems","description":"Natural Questions (NQ) benchmark containing real user questions issued to Google search with answers found from Wikipedia, designed for training and evaluation of automatic question answering systems","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Search","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/nq","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_af1a411ed6a2fdf0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/nq"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"nq","url":"https://llm-stats.com/benchmarks/nq","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","search","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_nqp-bench_0c5c1e38","familyId":"bmf_164c8f5b65ff","name":"NQP-Bench","oneLine":"NQP-Bench is a dataset within the OnePred paper for evaluating next-query prediction in multi-turn conversations. It spans three diverse subsets and is used to compare OnePred against baselines.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.23668","pdf":"https://arxiv.org/pdf/2605.23668","project":null,"code":"https://github.com/ZBWpro/OnePred","data":null,"hfPaper":"https://huggingface.co/papers/2605.23668"},"evidence":{"snippet":"To establish a rigorous testbed, we introduce NQP-Bench, spanning three diverse subsets.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.23668"},"ranking":{},"description":"NQP-Bench is a dataset within the OnePred paper for evaluating next-query prediction in multi-turn conversations. It spans three diverse subsets and is used to compare OnePred against baselines.","whyItMatters":"Next-query prediction lacks dedicated benchmarks; NQP-Bench provides a testbed for this task, enabling evaluation of proactive conversational systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"091f6a8b6836370140597e7d8a58fcbdf76e912f03af5ce6514329170c654c54"},"motivation":"Although large language model (LLM) conversational systems process millions of multi-turn dialogues daily, they remain fundamentally reactive: they respond only after the user types a query.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.23668","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_nrityam_dcb4ad43","familyId":"bmf_9db5bbb1c2e7","name":"NRITYAM","oneLine":"A dataset of 9,260 question-answer pairs across 12 languages evaluating cultural knowledge in global dance traditions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19727","pdf":"https://arxiv.org/pdf/2606.19727","project":null,"code":"https://github.com/niladrighosh03/NRITYAM","data":null,"hfPaper":"https://huggingface.co/papers/2606.19727"},"evidence":{"snippet":"To address this gap, we present NRITYAM, a comprehensive benchmark for evaluating the cultural comprehension capabilities of language models in the context of global dance traditions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19727"},"ranking":{"90d":{"score":28,"rank":280,"coverage":0.55,"confidence":"Low"}},"description":"A dataset of 9,260 question-answer pairs across 12 languages evaluating cultural knowledge in global dance traditions.","whyItMatters":"Addresses the gap in evaluating language models' cultural comprehension, particularly for traditional performing arts, offering a multilingual resource for assessing socio-cultural understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"80cb762032c6904e90d6ad47b59da00078bbf12971795372ff80c836ad9134d3"},"motivation":"Language models have become essential tools in shaping modern workflows.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19727","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_nrt-bench_4404866b","familyId":"bmf_cb5ba2557c72","name":"NRT-Bench","oneLine":"Evaluates multi-turn red-teaming of LLM agents acting as operators of a simulated nuclear power plant control room. Safety is measured by objective loss of critical safety functions (CSFs) under adaptive attacks.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.20408","pdf":"https://arxiv.org/pdf/2606.20408","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.20408"},"evidence":{"snippet":"We present NRT-Bench, a benchmark for multi-turn red-teaming of LLM agents acting as operators of a safety-critical system, instantiated in a simulated nuclear power plant control room.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.20408"},"ranking":{"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Evaluates multi-turn red-teaming of LLM agents acting as operators of a simulated nuclear power plant control room. Safety is measured by objective loss of critical safety functions (CSFs) under adaptive attacks.","whyItMatters":"Provides a repeatable environment for assessing LLM agent robustness in safety-critical operations, where failure is objective rather than judged, and supports reproducible safety evaluation across models and defences.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"82c84e9616ac2a4127906efffb1c2ea6a54fd9ee9b316bc9198a837587256640"},"motivation":"Large language model (LLM) agents are increasingly proposed as supervisory components for safety-critical systems, yet their robustness under sustained, adaptive adversarial pressure remains poorly characterized.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.20408","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_nuclearqav2_79073391","familyId":"bmf_30d04771464d","name":"NuclearQAv2","oneLine":"NuclearQAv2 evaluates LLMs on nuclear engineering knowledge with approximately 1,240 QA pairs covering boolean, numeric, and verbal questions. It uses structured prompting for automated question generation and response evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.27047","pdf":"https://arxiv.org/pdf/2606.27047","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.27047"},"evidence":{"snippet":"To address the need for systematic evaluation in this domain, we introduce NuclearQAv2, a benchmark for assessing LLMs on nuclear engineering knowledge.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.27047"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"NuclearQAv2 evaluates LLMs on nuclear engineering knowledge with approximately 1,240 QA pairs covering boolean, numeric, and verbal questions. It uses structured prompting for automated question generation and response evaluation.","whyItMatters":"Technical domains like nuclear engineering require reliable LLM evaluation. NuclearQAv2 provides a scalable benchmark to assess factual knowledge, quantitative reasoning, and conceptual understanding in this domain.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5538f6caecec97772f7e01820aab1d86b34a5dd871d9351c8dda55373eaa068b"},"motivation":"Large language models (LLMs) have demonstrated strong performance across a wide range of tasks, but ensuring their reliability in highly technical domains remains a significant challenge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.27047","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_numbench_cf80a48e","familyId":"bmf_f2a79fd9ccd2","name":"NumBench","oneLine":"Text-to-image (T2I) models often generate the wrong number of objects, yet existing benchmarks are too small or weakly controlled to explain why.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.28206v1","pdf":"https://arxiv.org/pdf/2608.28206v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce \\textbf{NumBench}, a benchmark of 640{,}000 prompts spanning 1{,}600 categories and counts from 1 to 100.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.28206"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"Text-to-image (T2I) models often generate the wrong number of objects, yet existing benchmarks are too small or weakly controlled to explain why.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T07:23:28.296315Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.28206v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_numerosityvlm-a-cognitively-inspired-bench_6f2495fb","familyId":"bmf_804131cc2b86","name":"NumerosityVLM","oneLine":"Evaluates zero-shot numerosity perception in vision-language models on 10,800 synthetic images across six controlled conditions manipulating object size, spatial arrangement, and numerosity while ablating texture, shape, and color.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-15","firstSeenAt":"2026-08-19","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.15425","pdf":"https://arxiv.org/pdf/2608.15425","project":null,"code":"https://github.com/fuy3/NumerosityVLM-Benchmark","data":"https://huggingface.co/datasets/fuy3/NumerosityVLM","hfPaper":null},"evidence":{"snippet":"We introduce a cognitively inspired diagnostic benchmark, NumerosityVLM, comprising 10,800 synthetic images across six controlled conditions.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":1773,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2608.15425"},"ranking":{"30d":{"score":18,"rank":164,"coverage":1.0,"confidence":"High","datasetDownloadRank":4,"datasetRankPopulation":30},"90d":{"score":24,"rank":303,"coverage":1.0,"confidence":"High","datasetDownloadRank":9,"datasetRankPopulation":66}},"description":"Evaluates zero-shot numerosity perception in vision-language models on 10,800 synthetic images across six controlled conditions manipulating object size, spatial arrangement, and numerosity while ablating texture, shape, and color.","whyItMatters":"Isolates numerosity from correlated visual features, enabling targeted diagnosis of number understanding in VLMs and revealing that architecture-driven language components limit performance.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"c7ae0054ca6e392a9d939120348a5436a91f35e41953ae9ccfa7556fa0bb7b82"},"motivation":"Vision-language models (VLMs) achieve strong performance on high-level multimodal tasks, yet numerosity perception, a cognitive ability that emerges in human infants before language acquisition, remains poorly understood in current models, as existing counting benchmarks entangle numerosity with correlated visual factors.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally named, provides code and data artifacts for reuse, and defines a fixed evaluation protocol with zero-shot scoring.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce a cognitively inspired diagnostic benchmark, NumerosityVLM, comprising 10,800 synthetic images across six controlled conditions."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15425","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":35,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses a niche but active area of vision-language model interpretability with public code and data, likely attracting moderate attention from cognitive AI researchers."},"evaluationMode":"public_reusable","publishers":[{"name":"fuy3","organizationType":"community","sourceUrl":"https://github.com/fuy3/NumerosityVLM-Benchmark","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_01ebb251e2f68e0e","familyId":"catalog_family_01ebb251e2f68e0e","name":"Nuscene","oneLine":"A multimodal benchmark for scene understanding and reasoning over the nuScenes autonomous driving domain.","description":"A multimodal benchmark for scene understanding and reasoning over the nuScenes autonomous driving domain.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Spatial","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/nuscene","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_01ebb251e2f68e0e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/nuscene"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"nuscene","url":"https://llm-stats.com/benchmarks/nuscene","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","spatial","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_oao-attackbench_8518a31c","familyId":"bmf_ebc7ad501506","name":"OAO-AttackBench","oneLine":"OAO-AttackBench is a set of counterfactual prompts for one-and-only objects, used to evaluate text-to-image alignment in a specific study.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.30262","pdf":"https://arxiv.org/pdf/2606.30262","project":"https://soyoun-won.github.io/one-and-only-ir-guidance/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.30262"},"evidence":{"snippet":"To systematically evaluate the challenging task of aligning generative outputs with unusual prompts for OAO objects, we introduce OAO-AttackBench, a benchmark comprising counterfactual prompts that directly conflict with the core visual identity of OAO objects.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.30262"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OAO-AttackBench is a set of counterfactual prompts for one-and-only objects, used to evaluate text-to-image alignment in a specific study.","whyItMatters":"The benchmark supports evaluating prompt adherence for concepts with strong visual priors, but it is introduced primarily to validate the proposed method and lacks a standalone public comparison path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"61373e2c58bd1fe4bbc74ed2b2e2d3c1f6d3d5864bbc12c1a1e72bfc57210870"},"motivation":"Text-to-image (T2I) diffusion models often fail to faithfully render explicit textual descriptions, instead defaulting to strongly learned visual priors due to a phenomenon referred to as concept association bias.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026","evidence":"Accepted at ECCV 2026","evidenceUrl":"https://arxiv.org/abs/2606.30262","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026","reviewStatus":"accepted","decisionRaw":"Accepted at ECCV 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.30262","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at ECCV 2026","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_96c6992e4a7c304c","familyId":"catalog_family_96c6992e4a7c304c","name":"Objectron","oneLine":"Objectron evaluates 3D object detection and pose estimation capabilities.","description":"Objectron evaluates 3D object detection and pose estimation capabilities.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["3D","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/objectron","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_96c6992e4a7c304c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/objectron"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"objectron","url":"https://llm-stats.com/benchmarks/objectron","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["3d","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_oblivion_83c0aaa0","familyId":"bmf_ea841fbbb09f","name":"OBLIVION","oneLine":"Evaluates operational skill unlearning in deployed agents across 88 attack episodes, measuring attack success rate and impact-weighted exposure after workflow-level defenses.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.08264","pdf":"https://arxiv.org/pdf/2608.08264","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.08264"},"evidence":{"snippet":"We introduce OBLIVION, a controlled benchmark and defense harness for revoked-skill resurrection.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08264"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates operational skill unlearning in deployed agents across 88 attack episodes, measuring attack success rate and impact-weighted exposure after workflow-level defenses.","whyItMatters":"Introduces a measurable benchmark for a new safety problem: preventing agents from rebuilding revoked skills, supporting workflow-level evaluation beyond parameter forgetting.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"c80ff3f8bf8837012becc8b8c6cfb5f4dbd569fe5211db4f27ada2a8a8565a31"},"motivation":"Large language model agents are becoming operational interfaces to files, memories, registries, and external tools.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.08264","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"OBLIVION Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.08264","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_obsdrivebench_13a5d40b","familyId":"bmf_75ddbe4da96d","name":"ObsDriveBench","oneLine":"ObsDriveBench evaluates multimodal understanding in autonomous driving under adverse weather, covering observability awareness, spatial reliability, and risk-aware decision-making with multiple-choice tasks over camera, LiDAR, and radar inputs.","area":"Multimodal","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Automotive"],"capabilities":[],"topics":["Multimodal"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-07-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.23537","pdf":"https://arxiv.org/pdf/2607.23537","project":null,"code":"https://github.com/russellyq/ObsDriveBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.23537"},"evidence":{"snippet":"To study this, we introduce \\textbf{ObsDriveBench}, a real-world multi-modal benchmark for adverse-weather autonomous driving.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23537"},"ranking":{"90d":{"score":31,"rank":235,"coverage":0.55,"confidence":"Low"}},"description":"ObsDriveBench evaluates multimodal understanding in autonomous driving under adverse weather, covering observability awareness, spatial reliability, and risk-aware decision-making with multiple-choice tasks over camera, LiDAR, and radar inputs.","whyItMatters":"Existing benchmarks rarely assess vision-language models under real-world adverse conditions with multimodal inputs. ObsDriveBench targets this gap by providing a fine-grained diagnosis of model behavior when observations are unreliable, supporting safer autonomous driving systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1b189325d747f14dcd6725eb7c971dae0b02064d2bff6eeecd592514a1f01262"},"motivation":"Autonomous driving under adverse weather remains a critical challenge, yet existing vision-language benchmarks mainly evaluate under standard conditions, synthetic corruptions, or single modality.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.23537","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_obshazard-bench_aa08033a","familyId":"bmf_697b42360627","name":"Obshazard-bench","oneLine":"Obshazard-bench is a real-time benchmark for disaster intelligence, integrating raw satellite and ground-station data. It covers 8 disaster categories and 28 sub-categories across 60+ countries, with VQA samples and a three-stage evaluation taxonomy: Predictive Crisis Anticipation, Active Evolution Reasoning, and Multi-faceted Impact Quantification.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00012","pdf":"https://arxiv.org/pdf/2608.00012","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00012"},"evidence":{"snippet":"To bridge this gap, we introduce Obshazard-bench, a real-time, observation-driven benchmark for evaluating disaster intelligence in MLLMs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00012"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Obshazard-bench is a real-time benchmark for disaster intelligence, integrating raw satellite and ground-station data. It covers 8 disaster categories and 28 sub-categories across 60+ countries, with VQA samples and a three-stage evaluation taxonomy: Predictive Crisis Anticipation, Active Evolution Reasoning, and Multi-faceted Impact Quantification.","whyItMatters":"Fills the gap in evaluating MLLMs for operational disaster response, which require real-time reasoning from raw observation streams. It provides a realistic testbed for assessing decision-support capabilities in evolving emergencies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"52071dd3e129915c8dee4b0baa8e5a5714fed0165eef5d197a01434a8343478e"},"motivation":"Multimodal Large Language Models (MLLMs) are increasingly used to interpret Earth observation data, yet their capability to support real-world disaster emergency response remains insufficiently evaluated.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00012","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_occur-bench_c8c0341e","familyId":"bmf_e63aa94e3c0e","name":"OCCUR-Bench","oneLine":"OCCUR-Bench evaluates temporal preservation in conversational image editing, providing occlusion-and-revelation scenarios with historical restoration references to assess faithful restoration of occluded content.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.07051","pdf":"https://arxiv.org/pdf/2607.07051","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.07051"},"evidence":{"snippet":"We introduce OCCUR-Bench, a diagnostic benchmark for temporal preservation in conversational image editing.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.07051"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OCCUR-Bench evaluates temporal preservation in conversational image editing, providing occlusion-and-revelation scenarios with historical restoration references to assess faithful restoration of occluded content.","whyItMatters":"Targets the under-explored problem of preserving content that temporarily disappears during multi-turn editing, offering a diagnostic benchmark for faithful restoration.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"733c62fcbb9583195c9205d3a59310c787577327fb0281b032ac1bdd23fbc8bc"},"motivation":"Conversational image editing requires preserving not only visible content, but also content that temporarily disappears across turns.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.07051","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_917413cfda40d9db","familyId":"catalog_family_917413cfda40d9db","name":"OCRBench","oneLine":"OCRBench: Comprehensive evaluation benchmark for assessing Optical Character Recognition (OCR) capabilities in Large Multimodal Models across text recognition, scene text VQA, and document understanding tasks","description":"OCRBench: Comprehensive evaluation benchmark for assessing Optical Character Recognition (OCR) capabilities in Large Multimodal Models across text recognition, scene text VQA, and document understanding tasks","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Image To Text","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ocrbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_917413cfda40d9db"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ocrbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ocrbench","url":"https://llm-stats.com/benchmarks/ocrbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image to text","vision"],"catalogModelCount":24,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_382d6086e1c5a6a9","familyId":"catalog_family_382d6086e1c5a6a9","name":"OCRBench V2","oneLine":"OCRBench v2: Enhanced large-scale bilingual benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with 10,000 human-verified question-answering pairs across 8 core OCR capabilities","description":"OCRBench v2: Enhanced large-scale bilingual benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with 10,000 human-verified question-answering pairs across 8 core OCR capabilities","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Image To Text","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2501.00321","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_382d6086e1c5a6a9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/ocrbenchv2"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ocrbench-v2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"ocrBenchV2","url":"https://benchlm.ai/benchmarks/ocrbenchv2","paperUrl":"https://arxiv.org/abs/2501.00321","year":"2025","fullName":"OCRBench V2","format":"Accuracy","tasks":"Image OCR tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"ocrbench-v2","url":"https://llm-stats.com/benchmarks/ocrbench-v2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","image to text","vision"],"catalogModelCount":7,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_3a2efe4fa70bff40","familyId":"catalog_family_3a2efe4fa70bff40","name":"OCRBench-V2 (en)","oneLine":"OCRBench v2 English subset: Enhanced benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with English text content","description":"OCRBench v2 English subset: Enhanced benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with English text content","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Image To Text","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ocrbench-v2-(en)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3a2efe4fa70bff40"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ocrbench-v2-(en)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ocrbench-v2-(en)","url":"https://llm-stats.com/benchmarks/ocrbench-v2-(en)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image to text","vision"],"catalogModelCount":14,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_058b7797debf92d2","familyId":"catalog_family_058b7797debf92d2","name":"OCRBench-V2 (zh)","oneLine":"OCRBench v2 Chinese subset: Enhanced benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with Chinese text content","description":"OCRBench v2 Chinese subset: Enhanced benchmark for evaluating Large Multimodal Models on visual text localization and reasoning with Chinese text content","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Image To Text","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ocrbench-v2-(zh)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_058b7797debf92d2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ocrbench-v2-(zh)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ocrbench-v2-(zh)","url":"https://llm-stats.com/benchmarks/ocrbench-v2-(zh)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image to text","vision"],"catalogModelCount":11,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_oct-bench_3914e5be","familyId":"bmf_b657b809f328","name":"OCT-Bench","oneLine":"OCT-Bench evaluates multimodal large language models on OCT image understanding with 10,076 multiple-choice questions from 4,137 images across seven public datasets, covering 20 tasks in Perception, Cognition, and Reasoning.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.16609","pdf":"https://arxiv.org/pdf/2607.16609","project":null,"code":"https://github.com/baochenfu/OCT-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.16609"},"evidence":{"snippet":"To address this limitation, we introduce OCT-Bench, a comprehensive benchmark dedicated to OCT image understanding.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":null,"githubStars":162,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.16609"},"ranking":{"90d":{"score":59,"rank":22,"coverage":0.7,"confidence":"Medium"}},"description":"OCT-Bench evaluates multimodal large language models on OCT image understanding with 10,076 multiple-choice questions from 4,137 images across seven public datasets, covering 20 tasks in Perception, Cognition, and Reasoning.","whyItMatters":"Offers a comprehensive benchmark for OCT understanding beyond coarse classification, enabling capability bottleneck analysis for clinical deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"238c1136dffdc0b4d1084bc66bcb501e556b225bc0a4c86211000e3b45dc0ad4"},"motivation":"Optical coherence tomography (OCT) imaging is essential for the diagnosis and treatment of retinal diseases.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.16609","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"baochenfu","organizationType":"community","sourceUrl":"https://github.com/baochenfu/OCT-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_e4a8cc0a8b5ac334","familyId":"catalog_family_e4a8cc0a8b5ac334","name":"OctoCodingBench","oneLine":"Octopus coding benchmark for evaluating multi-language programming capabilities","description":"Octopus coding benchmark for evaluating multi-language programming capabilities","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/octocodingbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e4a8cc0a8b5ac334"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/octocodingbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"octocodingbench","url":"https://llm-stats.com/benchmarks/octocodingbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_odineval_a7103a7e","familyId":"bmf_3047befaf9e9","name":"OdinEval","oneLine":"OdinEval is a benchmark for program repair in the Odin programming language, built from documented defects, with issue-to-commit bindings, regression tests, and execution records.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation"],"topics":["cs.SE"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-19","firstSeenAt":"2026-08-20","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.18595","pdf":"https://arxiv.org/pdf/2608.18595","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present OdinEval, a reproducible benchmark built from documented defects in public Odin repositories.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.18595"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OdinEval is a benchmark for program repair in the Odin programming language, built from documented defects, with issue-to-commit bindings, regression tests, and execution records.","whyItMatters":"Existing repair benchmarks focus on mainstream languages, leaving systems languages like Odin untested. OdinEval provides a reproducible benchmark for a less-covered language.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6618ab0cd023268560fb59d1e91e20569a6dddb7e88ba6a6565cae82d884a0dd"},"motivation":"Repository-level repair benchmarks still center on a few mainstream languages, leaving systems languages such as Odin largely untested.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-20T09:45:58.504563Z","model":"deepseek-v4-flash","decisionReason":"Named benchmark with clear release artifacts and reproducible protocol, including frozen data and containers."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.18595","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-20T09:44:47.308583Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_8cd96084e50818fb","familyId":"catalog_family_8cd96084e50818fb","name":"ODinW","oneLine":"Object Detection in the Wild (ODinW) benchmark for evaluating object detection models' task-level transfer ability across diverse real-world datasets in terms of prediction accuracy and adaptation efficiency","description":"Object Detection in the Wild (ODinW) benchmark for evaluating object detection models' task-level transfer ability across diverse real-world datasets in terms of prediction accuracy and adaptation efficiency","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/odinw","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8cd96084e50818fb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/odinw"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"odinw","url":"https://llm-stats.com/benchmarks/odinw","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["vision"],"catalogModelCount":16,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_aac5f1f80273bc77","familyId":"catalog_family_aac5f1f80273bc77","name":"ODINW13","oneLine":"A visual detection and grounding benchmark slice used to compare zero-shot object understanding across diverse domains.","description":"A visual detection and grounding benchmark slice used to compare zero-shot object understanding across diverse domains.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_aac5f1f80273bc77"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/odinw13"}],"catalogSources":[{"catalog":"benchlm","sourceId":"odinw13","url":"https://benchlm.ai/benchmarks/odinw13","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"ODINW13","format":"Detection and grounding","tasks":"Out-of-distribution object understanding","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_oeis-open_86208c3e","familyId":"bmf_5120c16b457f","name":"OEIS Open","oneLine":"Evaluates language models on formalized open mathematical conjectures from OEIS, scored by theorem resolution under a compute budget.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":0.8,"links":{"report":"https://arxiv.org/abs/2608.11941","pdf":"https://arxiv.org/pdf/2608.11941","project":null,"code":"https://github.com/epoch-research/LeanOpenProblems","data":null,"hfPaper":null},"evidence":{"snippet":"We construct OEIS Open, a benchmark based on 492 open mathematical conjectures from the OEIS, formalized in Lean by Tsoukalas et al.","reasonCodes":["exact coined title identity tied to benchmark evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11941"},"ranking":{"30d":{"score":37,"rank":56,"coverage":0.55,"confidence":"Low"},"90d":{"score":36,"rank":184,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates language models on formalized open mathematical conjectures from OEIS, scored by theorem resolution under a compute budget.","whyItMatters":"It provides a secure, reproducible suite for testing autonomous mathematical reasoning on open problems.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"539f099b28cbb32ccb0eb36c086c58837faeea7535a33600f7a3cd9bb149d408"},"motivation":"We construct OEIS Open, a benchmark based on 492 open mathematical conjectures from the OEIS, formalized in Lean by Tsoukalas et al.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"OEIS Open is a named benchmark with open-source evaluation code and results repositories, enabling reuse by other teams.","canonicalNameSource":"abstract","canonicalNameEvidence":"We construct OEIS Open, a benchmark based on 492 open mathematical conjectures"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11941","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":68,"confidence":"Low","horizon":"7d","reason":"The benchmark targets the niche area of formalized math conjectures with modest language model capabilities, though specific adoption signals are absent."},"evaluationMode":"public_reusable","publishers":[{"name":"Epoch AI","organizationType":"company-research-lab","sourceUrl":"https://github.com/epoch-research/LeanOpenProblems","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_oenobench_8cfa7d28","familyId":"bmf_0fd06eb77b86","name":"OenoBench","oneLine":"Evaluates LLM knowledge in the wine domain via 3,266 multiple-choice questions across six pillars and four difficulty tiers, sourced from verified facts and scored against a calibrated audit.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.20106","pdf":"https://arxiv.org/pdf/2608.20106","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce OenoBench, a wine-domain knowledge benchmark of 3,266 multiple-choice questions across six pillars (regions, grape varieties, viticulture, winemaking, producers, business) and four difficulty tiers.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.20106"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLM knowledge in the wine domain via 3,266 multiple-choice questions across six pillars and four difficulty tiers, sourced from verified facts and scored against a calibrated audit.","whyItMatters":"Provides a detailed, provenance-grounded benchmark for specialized knowledge, enabling comparison of LLM factual accuracy in a domain with clear source verification.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"d2b72f1ef7904905438136e62f540e1ca5fa3a9cbc09d249db5a486b59d61c19"},"motivation":"We introduce OenoBench, a wine-domain knowledge benchmark of 3,266 multiple-choice questions across six pillars (regions, grape varieties, viticulture, winemaking, producers, business) and four difficulty tiers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The paper declares a named benchmark, describes a stable scoring protocol, and explicitly states release of corpus, audit findings, and code, though no artifact links are provided in the input.","canonicalNameSource":"paper_title","canonicalNameEvidence":"OenoBench: A Wine-Domain Benchmark for Knowledge-Grounded Evaluation of Large Language Models"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.20106","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:50.392548Z"},"attentionForecast":{"score":55,"confidence":"Low","horizon":"7d","reason":"The unusual wine domain and large annotated dataset may attract niche but engaged interest, though missing direct artifact links could dampen immediate uptake."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_1eabad48f477d487","familyId":"catalog_family_1eabad48f477d487","name":"OfficeQA","oneLine":"Grounded numerical reasoning over a corpus of historical U.S. Treasury Bulletin documents.","description":"Grounded numerical reasoning over a corpus of historical U.S. Treasury Bulletin documents.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1eabad48f477d487"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/officeqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"officeQa","url":"https://benchlm.ai/benchmarks/officeqa","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"OfficeQA","format":"Agentic grounded QA accuracy","tasks":"Historical Treasury Bulletin questions","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_1b5ee4336ea31196","familyId":"catalog_family_1b5ee4336ea31196","name":"OfficeQA Pro","oneLine":"OfficeQA Pro evaluates AI models on professional knowledge-work questions and tasks drawn from real office workflows, including document analysis, spreadsheet reasoning, and information synthesis across business domains.","description":"OfficeQA Pro evaluates AI models on professional knowledge-work questions and tasks drawn from real office workflows, including document analysis, spreadsheet reasoning, and information synthesis across business domains.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Reasoning","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2603.08655","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1b5ee4336ea31196"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/officeqapro"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/officeqa-pro"}],"catalogSources":[{"catalog":"benchlm","sourceId":"officeQaPro","url":"https://benchlm.ai/benchmarks/officeqapro","paperUrl":"https://arxiv.org/abs/2603.08655","year":"2026","fullName":"OfficeQA Pro","format":"Grounded QA over office artifacts","tasks":"Document and spreadsheet tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"officeqa-pro","url":"https://llm-stats.com/benchmarks/officeqa-pro","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","reasoning","general","agents"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_offnadirloc_b7680d27","familyId":"bmf_eb7ecdfb3ed6","name":"OffNadirLoc","oneLine":"OffNadirLoc evaluates UAV-to-satellite geo-localization under large off-nadir views, with structure-aware contextual weighting and view-coherent learning for viewpoint-invariant features.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19951","pdf":"https://arxiv.org/pdf/2607.19951","project":"https://montalario.github.io/offnadirloc/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19951"},"evidence":{"snippet":"In this work, we introduce OffNadirLoc, a new benchmark for large off-nadir UAV-to-satellite geo-localization.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19951"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OffNadirLoc evaluates UAV-to-satellite geo-localization under large off-nadir views, with structure-aware contextual weighting and view-coherent learning for viewpoint-invariant features.","whyItMatters":"Existing benchmarks focus on near-nadir views, limiting real-world deployment. A dedicated benchmark improves evaluation of large off-nadir scenarios with structural understanding and multi-view consistency.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cc0c6ac07c4c898f09b02d71c914f429186c66cefe5f77931db49b72bee18589"},"motivation":"Cross-view geo-localization between UAV and satellite imagery remains a fundamental yet highly challenging task, especially under large off-nadir views where drastic perspective distortions, occlusions, and appearance gaps occur.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19951","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"OffNadirLoc project","organizationType":"academic-lab","sourceUrl":"https://montalario.github.io/offnadirloc/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_d3a0b0c6ea1d490b","familyId":"catalog_family_d3a0b0c6ea1d490b","name":"OJBench","oneLine":"OJBench is a competition-level code benchmark designed to assess the competitive-level code reasoning abilities of large language models. It comprises 232 programming competition problems from NOI and ICPC, categorized into Easy, Medium, and Hard difficulty levels. The benchmark evaluates models' ability to solve complex competitive programming challenges using Python and C++.","description":"OJBench is a competition-level code benchmark designed to assess the competitive-level code reasoning abilities of large language models. It comprises 232 programming competition problems from NOI and ICPC, categorized into Easy, Medium, and Hard difficulty levels. The benchmark evaluates models' ability to solve complex competitive programming challenges using Python and C++.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ojbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d3a0b0c6ea1d490b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ojbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ojbench","url":"https://llm-stats.com/benchmarks/ojbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":9,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_555e52f92861743e","familyId":"catalog_family_555e52f92861743e","name":"OJBench (C++)","oneLine":"OJBench (C++) is the C++ subset of OJBench, a competition-level code benchmark that evaluates large language models on programming competition problems from NOI and ICPC using C++ as the implementation language.","description":"OJBench (C++) is the C++ subset of OJBench, a competition-level code benchmark that evaluates large language models on programming competition problems from NOI and ICPC using C++ as the implementation language.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ojbench-cpp","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_555e52f92861743e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ojbench-cpp"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ojbench-cpp","url":"https://llm-stats.com/benchmarks/ojbench-cpp","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_38c6c428d892c7b6","familyId":"catalog_family_38c6c428d892c7b6","name":"olmOCR","oneLine":"An end-to-end document understanding benchmark over long, layout-rich PDFs with tables, equations, headers, footnotes, and multi-column flows.","description":"An end-to-end document understanding benchmark over long, layout-rich PDFs with tables, equations, headers, footnotes, and multi-column flows.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://github.com/allenai/olmocr/tree/main/olmocr/bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_38c6c428d892c7b6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/olmocr"}],"catalogSources":[{"catalog":"benchlm","sourceId":"olmOcr","url":"https://benchlm.ai/benchmarks/olmocr","paperUrl":"https://github.com/allenai/olmocr/tree/main/olmocr/bench","year":"2025","fullName":"olmOCR-Bench","format":"Mean accuracy","tasks":"Layout-rich PDF understanding","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_4af0fe63f717ff2f","familyId":"catalog_family_4af0fe63f717ff2f","name":"OlympiadBench","oneLine":"A challenging benchmark for promoting AGI with Olympiad-level bilingual multimodal scientific problems. Comprises 8,476 math and physics problems from international and Chinese Olympiads and the Chinese college entrance exam, featuring expert-level annotations for step-by-step reasoning. Includes both text-only and multimodal problems in English and Chinese.","description":"A challenging benchmark for promoting AGI with Olympiad-level bilingual multimodal scientific problems. Comprises 8,476 math and physics problems from international and Chinese Olympiads and the Chinese college entrance exam, featuring expert-level annotations for step-by-step reasoning. Includes both text-only and multimodal problems in English and Chinese.","area":"Mathematical Reasoning","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Math","Multimodal","Physics","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/olympiadbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4af0fe63f717ff2f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/olympiadbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"olympiadbench","url":"https://llm-stats.com/benchmarks/olympiadbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","multimodal","physics","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"bm_omegause-officeval_6b76ef98","familyId":"bmf_7046d745096e","name":"OmegaUse-OfficeVal","oneLine":"Evaluates LLM agents on long-horizon office-suite tasks from 100 practitioner-derived scenarios, scoring deliverable quality through code-based verifiers and comparing performance against economic signals of human labor time and price.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27155","pdf":"https://arxiv.org/pdf/2607.27155","project":"https://omegause-officeval.github.io","code":"https://github.com/baidu-frontier-research/OmegaUse-OfficeVal","data":null,"hfPaper":"https://huggingface.co/papers/2607.27155"},"evidence":{"snippet":"We introduce OmegaUse-OfficeVal, a benchmark for evaluating LLM agents on long-horizon office-suite tasks with task-level economic grounding.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":13,"hfDailySubmittedAt":"2026-07-30T00:00:00.000Z","githubStars":9,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27155"},"ranking":{"90d":{"score":39,"rank":157,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates LLM agents on long-horizon office-suite tasks from 100 practitioner-derived scenarios, scoring deliverable quality through code-based verifiers and comparing performance against economic signals of human labor time and price.","whyItMatters":"Existing agent benchmarks rarely consider cost-effectiveness for real office workflows. This benchmark provides a way to compare agent output quality relative to human labor costs, supporting value-weighted evaluation of office automation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2bad570f995c8afbc6c349d2793cf84552615b7dc98214ae191a25d8c28c6e43"},"motivation":"Large language model (LLM) agents are increasingly expected to assist users in completing tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27155","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Baidu Frontier Research","organizationType":"company-research-lab","sourceUrl":"https://github.com/baidu-frontier-research/OmegaUse-OfficeVal","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_omniabench_e9e6c4ef","familyId":"bmf_6ab9ee1bd426","name":"OmniaBench","oneLine":"The evaluation object is unclear from the provided information.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14989","pdf":"https://arxiv.org/pdf/2607.14989","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.14989"},"evidence":{"snippet":"We introduce OmniaBench, a benchmark for evaluating general agents across diverse scenarios with explicit state spaces.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14989"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"The evaluation object is unclear from the provided information.","whyItMatters":"The evaluation gap and practical decision value are unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7321f5df32355b8eef13cea8752328e05bfff2efcd49f269f6958cbab05ea239"},"motivation":"Large language models are increasingly evolving from text generators into general agents capable of understanding user requests, invoking external tools, and completing complex tasks through interaction.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14989","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_09205a714e6cee54","familyId":"catalog_family_09205a714e6cee54","name":"OmniBench","oneLine":"A novel multimodal benchmark designed to evaluate large language models' ability to recognize, interpret, and reason across visual, acoustic, and textual inputs simultaneously. Comprises 1,142 question-answer pairs covering 8 task categories from basic perception to complex inference, with a unique constraint that accurate responses require integrated understanding of all three modalities.","description":"A novel multimodal benchmark designed to evaluate large language models' ability to recognize, interpret, and reason across visual, acoustic, and textual inputs simultaneously. Comprises 1,142 question-answer pairs covering 8 task categories from basic perception to complex inference, with a unique constraint that accurate responses require integrated understanding of all three modalities.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/omnibench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_09205a714e6cee54"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/omnibench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"omnibench","url":"https://llm-stats.com/benchmarks/omnibench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_ddc7cf62e6710c6b","familyId":"catalog_family_ddc7cf62e6710c6b","name":"OmniBench Music","oneLine":"Music component of OmniBench, a comprehensive benchmark for evaluating omni-language models' ability to recognize, interpret, and reason across visual, acoustic, and textual inputs simultaneously. The music category includes various compositions and performances that require integrated understanding across text, image, and audio modalities.","description":"Music component of OmniBench, a comprehensive benchmark for evaluating omni-language models' ability to recognize, interpret, and reason across visual, acoustic, and textual inputs simultaneously. The music category includes various compositions and performances that require integrated understanding across text, image, and audio modalities.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/omnibench-music","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ddc7cf62e6710c6b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/omnibench-music"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"omnibench-music","url":"https://llm-stats.com/benchmarks/omnibench-music","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","audio"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_omnicad_d45eb5cc","familyId":"bmf_25a3cde22ce0","name":"OmniCAD","oneLine":"Evaluates VLMs on assembly-aware 3D spatial reasoning across 25k mechanical assemblies with component-level, relational, and tool-augmented tasks.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning","Geometric reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.22637v1","pdf":"https://arxiv.org/pdf/2608.22637v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce OmniCAD, a large-scale benchmark for assembly-aware 3D spatial reasoning across diverse industrial systems, including robotic mechanisms, automotive components, aerospace structures, and agricultural machinery.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22637"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates VLMs on assembly-aware 3D spatial reasoning across 25k mechanical assemblies with component-level, relational, and tool-augmented tasks.","whyItMatters":"Industrial assembly reasoning requires physically valid 3D understanding, and this benchmark exposes current VLM limitations in that domain.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"5a527e53f6df67784fbd73d60741b4e2751cc19064b762669bc43eb13b6fcecd"},"motivation":"Recent vision-language models (VLMs) show strong capabilities in robotic perception and spatial reasoning, yet their ability to reason about complex mechanical assemblies remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is large-scale with defined tasks and will be open-sourced, providing a clear public reuse path.","canonicalNameSource":"paper_title","canonicalNameEvidence":"OmniCAD: A Large-Scale Benchmark for 3D Spatial Reasoning in Robotics Assemblies"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22637v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"Robotics and 3D spatial reasoning are trending topics, and the scale and industrial grounding of OmniCAD may attract significant attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_omnicap-if_98c29629","familyId":"bmf_3e95cd9aa687","name":"OmniCap-IF","oneLine":"OmniCap-IF evaluates instruction following in omni-modal video captioning with 50 constraint types, 1,920 samples, and checklist-based scoring for format and content correctness.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08572","pdf":"https://arxiv.org/pdf/2606.08572","project":null,"code":"https://github.com/NJU-LINK/omnicap-if","data":null,"hfPaper":"https://huggingface.co/papers/2606.08572"},"evidence":{"snippet":"To bridge this gap, we introduce OmniCap-IF, the first comprehensive benchmark specifically designed to evaluate instruction-following capabilities in omni-modal captioning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":14,"hfDailySubmittedAt":"2026-06-09T00:00:00.000Z","githubStars":11,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08572"},"ranking":{"90d":{"score":41,"rank":143,"coverage":0.7,"confidence":"Medium"}},"description":"OmniCap-IF evaluates instruction following in omni-modal video captioning with 50 constraint types, 1,920 samples, and checklist-based scoring for format and content correctness.","whyItMatters":"Existing benchmarks miss the interplay of audio-visual and user constraints; OmniCap-IF provides a fine-grained evaluation to expose format-content tradeoffs and drive improvements.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8c8ffb534a25992679278a1e0d6143005bf114ca3eed4f0946c1a9b4ef0d8164"},"motivation":"While Omni-modal Large Language Models (OLLMs) have demonstrated impressive capabilities in jointly processing audio and visual streams, their ability to strictly adhere to complex, multi-faceted user instructions remains largely unexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08572","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"NJU-LINK","organizationType":"academic-lab","sourceUrl":"https://github.com/NJU-LINK/omnicap-if","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_1685ef9a1aefe280","familyId":"catalog_family_1685ef9a1aefe280","name":"OmniDocBench","oneLine":"OmniDocBench evaluates multimodal models on document understanding tasks such as OCR, layout parsing, and structured document comprehension.","description":"OmniDocBench evaluates multimodal models on document understanding tasks such as OCR, layout parsing, and structured document comprehension.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Document Understanding","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1685ef9a1aefe280"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/omnidocbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/omnidocbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"omniDocBench","url":"https://benchlm.ai/benchmarks/omnidocbench","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"OmniDocBench","format":"Document-understanding score","tasks":"Complex document-understanding tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"omnidocbench","url":"https://llm-stats.com/benchmarks/omnidocbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","document understanding","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_f49c6a408192e277","familyId":"catalog_family_f49c6a408192e277","name":"OmniDocBench 1.5","oneLine":"OmniDocBench 1.5 is a comprehensive benchmark for evaluating multimodal large language models on document understanding tasks, including OCR, document parsing, information extraction, and visual question answering across diverse document types. Lower Overall Edit Distance scores are better.","description":"OmniDocBench 1.5 is a comprehensive benchmark for evaluating multimodal large language models on document understanding tasks, including OCR, document parsing, information extraction, and visual question answering across diverse document types. Lower Overall Edit Distance scores are better.","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Reasoning","Structured Output","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f49c6a408192e277"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/omnidocbench15"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/omnidocbench-1.5"}],"catalogSources":[{"catalog":"benchlm","sourceId":"omniDocBench15","url":"https://benchlm.ai/benchmarks/omnidocbench15","paperUrl":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","year":"2026","fullName":"OmniDocBench 1.5","format":"Document understanding benchmark","tasks":"Document understanding tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"omnidocbench-1.5","url":"https://llm-stats.com/benchmarks/omnidocbench-1.5","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","structured output","vision"],"catalogModelCount":18,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general"},{"id":"bm_omniedit-bench_6a691145","familyId":"bmf_48dbdad242c2","name":"OmniEdit-Bench","oneLine":"OmniEdit-Bench evaluates instruction-based video editing with dimensions including spatial, temporal, audio, and reference-based editing, and assesses accuracy, preservation, realism, and consistency with an accuracy-aware penalty.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.05049","pdf":"https://arxiv.org/pdf/2608.05049","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.05049"},"evidence":{"snippet":"To address these issues, we introduce a comprehensive and structured benchmark for IVE.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.05049"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"OmniEdit-Bench evaluates instruction-based video editing with dimensions including spatial, temporal, audio, and reference-based editing, and assesses accuracy, preservation, realism, and consistency with an accuracy-aware penalty.","whyItMatters":"Existing video editing benchmarks have limited task coverage and metrics that fail to measure instruction fidelity. OmniEdit-Bench provides a structured evaluation that prevents incorrect edits from receiving inflated scores.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"83b795e2383ec191214c586db1be83d6d36a86f7c73504bd0cfe4dc63ea28595"},"motivation":"Instruction-based video editing (IVE) is an emerging field with broad applications, yet evaluating editing models remains challenging.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.05049","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_omnieeg-bench_5de66c8c","familyId":"bmf_16b11959715a","name":"OmniEEG-Bench","oneLine":"OmniEEG-Bench is a unified benchmark for EEG foundation models, organizing evaluation into six task families and unifying 54 EEG datasets. It provides a leaderboard and code repository.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00815","pdf":"https://arxiv.org/pdf/2606.00815","project":null,"code":"https://github.com/ncclab-sustech/omni-eegbench.git","data":null,"hfPaper":"https://huggingface.co/papers/2606.00815"},"evidence":{"snippet":"Here, we introduce OmniEEG-Bench, a unified benchmark and downstream task roadmap for EEG foundation models (FMs).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":19,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00815"},"ranking":{},"description":"OmniEEG-Bench is a unified benchmark for EEG foundation models, organizing evaluation into six task families and unifying 54 EEG datasets. It provides a leaderboard and code repository.","whyItMatters":"Standardizes evaluation of EEG foundation models across diverse tasks, revealing scaling-law behavior and the importance of pretraining data diversity.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2b7ccf2ce1d3021c9c38a8d157a4da2d12134a41c3aa1797a6693a87761fbe24"},"motivation":"Electroencephalography (EEG) supports a variety of brain-computer interface (BCI) tasks ranging from brain-state monitoring to human-LLM interactions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00815","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_omnifood-bench_13bf03fa","familyId":"bmf_11629f46fc0a","name":"OmniFood-Bench","oneLine":"A benchmark for evaluating Vision-Language Models on nutrient reasoning and personalized health advice, with progressive capabilities: basic perception, quantitative reasoning, and safety-critical advisory.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.08423","pdf":"https://arxiv.org/pdf/2607.08423","project":"https://anonymous.4open.science/r/OmniFood-Bench-7D0B","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.08423"},"evidence":{"snippet":"In this paper, we introduce OmniFood-Bench, a comprehensive benchmark constructed from the MM-Food-100K dataset.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.08423"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark for evaluating Vision-Language Models on nutrient reasoning and personalized health advice, with progressive capabilities: basic perception, quantitative reasoning, and safety-critical advisory.","whyItMatters":"The benchmark targets a critical gap in food systems AI evaluation, which often focuses on classification. By testing reasoning to safety-critical advice, it aims to establish standards for trustworthiness in public health applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e7ca01a5a8a80c7c669b4edbfae234abc7e00a4aa70688a205bea2131cd03686"},"motivation":"The rapid integration of Large Vision-Language Models (VLMs) into critical infrastructure promises to revolutionize personalized healthcare and dietary management.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.08423","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_1589f33962d88a92","familyId":"catalog_family_1589f33962d88a92","name":"OmniGAIA","oneLine":"OmniGAIA evaluates multimodal perception and reasoning in agentic contexts, testing a model's ability to process diverse inputs and perform complex multi-step reasoning tasks.","description":"OmniGAIA evaluates multimodal perception and reasoning in agentic contexts, testing a model's ability to process diverse inputs and perform complex multi-step reasoning tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/omnigaia","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1589f33962d88a92"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/omnigaia"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"omnigaia","url":"https://llm-stats.com/benchmarks/omnigaia","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_omnigamearena_6afe515a","familyId":"bmf_bdedbd5c5172","name":"OmniGameArena","oneLine":"OmniGameArena evaluates VLM game agents across twelve Unreal Engine 5 games spanning Solo, PvP, and Coop play. It provides unified action interfaces and two evaluation clocks (PDQ for decision quality, LCRT for real-time latency). The Improvement Dynamics Curve (IDC) measures agent improvement through reflection rounds.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.09826","pdf":"https://arxiv.org/pdf/2606.09826","project":null,"code":"https://github.com/mxlin043/OmniGameArena","data":null,"hfPaper":"https://huggingface.co/papers/2606.09826"},"evidence":{"snippet":"We address these gaps with OmniGameArena, a real-time benchmark of twelve newly built Unreal Engine 5 games spanning Solo (7), PvP (3), and Coop (2) with unified action interfaces, and the Improvement Dynamics Curve (IDC), an agentic-reflection harness in which a tool-using reflector LLM autonomously refines a bounded skill prompt across multiple rounds.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":19,"hfDailySubmittedAt":"2026-06-09T00:00:00.000Z","githubStars":41,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09826"},"ranking":{"90d":{"score":51,"rank":58,"coverage":0.7,"confidence":"Medium"}},"description":"OmniGameArena evaluates VLM game agents across twelve Unreal Engine 5 games spanning Solo, PvP, and Coop play. It provides unified action interfaces and two evaluation clocks (PDQ for decision quality, LCRT for real-time latency). The Improvement Dynamics Curve (IDC) measures agent improvement through reflection rounds.","whyItMatters":"Existing game benchmarks report single scores and lack unified protocols for heterogeneous agents. OmniGameArena offers a real-time benchmark with multiple observables, including improvement dynamics, facilitating comparative evaluation of different agent classes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a05d9d2d817c0f5f2799299c7ebb66c137ba23d2cefb0bb95254e2e1a85a137e"},"motivation":"Vision-language model (VLM) agents are increasingly deployed in interactive game environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09826","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"OmniGameArena Team","organizationType":"academic-lab","sourceUrl":"https://github.com/mxlin043/OmniGameArena","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_omnihandwritingocr_85cf42ca","familyId":"bmf_485481a48f61","name":"OmniHandwritingOCR","oneLine":"Evaluates multimodal LLMs and OCR systems on handwritten text and mathematical expression recognition across six subtasks and twelve subsets with 77.57K images.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-19","firstSeenAt":"2026-08-20","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.18586","pdf":"https://arxiv.org/pdf/2608.18586","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce OmniHandwritingOCR, a diagnostic benchmark for evaluating MLLMs and OCR systems on handwritten OCR.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.18586"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates multimodal LLMs and OCR systems on handwritten text and mathematical expression recognition across six subtasks and twelve subsets with 77.57K images.","whyItMatters":"Provides a challenging diagnostic benchmark for realistic handwritten OCR, highlighting failures in complex formulas and visual grounding that existing printed-text benchmarks miss.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"a13f277371086f1142d4638e8fe2044697e25c4aef8f50f31e9d1449ea6a480b"},"motivation":"Multimodal large language models (MLLMs) are increasingly used as OCR systems in document and knowledge-processing pipelines, but their ability to faithfully read real handwriting remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"The paper defines a specific benchmark with subsets, metrics, and a unified protocol, and the arxiv report serves as a public source for the evaluation contract.","canonicalNameSource":"paper_title","canonicalNameEvidence":"OmniHandwritingOCR: A Diagnostic Benchmark for Evaluating Multimodal LLMs in Handwritten OCR Scenarios"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.18586","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-20T14:20:56.750721Z"},"attentionForecast":{"score":60,"confidence":"Medium","horizon":"7d","reason":"The CIKM venue and comprehensive handwritten OCR coverage, including new formula corpus, may draw interest from document AI researchers."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_omniinteract_26f9ce32","familyId":"bmf_2f71b445271d","name":"OmniInteract","oneLine":"OmniInteract is a streaming benchmark for real-time omnimodal LLMs evaluated through native online inference over audio-visual streams. It contains 250 videos with 1,430 temporally grounded response slots (1Q1A and 1QnA), with each slot including trigger, response window, and target answer. Metrics include IA-QTF1, Interruption Diagnostic Suite, and Nested Chain Completion Score.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26485","pdf":"https://arxiv.org/pdf/2605.26485","project":null,"code":"https://github.com/Lucky-Lance/OmniInteract","data":null,"hfPaper":"https://huggingface.co/papers/2605.26485"},"evidence":{"snippet":"We introduce OmniInteract, a streaming benchmark for real-time omnimodal large language models evaluated through native online inference over audio-visual streams.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":null,"githubStars":20,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26485"},"ranking":{},"description":"OmniInteract is a streaming benchmark for real-time omnimodal LLMs evaluated through native online inference over audio-visual streams. It contains 250 videos with 1,430 temporally grounded response slots (1Q1A and 1QnA), with each slot including trigger, response window, and target answer. Metrics include IA-QTF1, Interruption Diagnostic Suite, and Nested Chain Completion Score.","whyItMatters":"Real-time omnimodal assistants must process streaming audio-visual input and decide whether and when to respond without access to future content, a capability not captured by offline video benchmarks. OmniInteract provides a native streaming evaluation protocol, revealing that current models remain weak in streaming interaction.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5359f57021ec3668aa29ca67f89097c943ae5ab02aefe65d66d7c54154e74079"},"motivation":"We introduce OmniInteract, a streaming benchmark for real-time omnimodal large language models evaluated through native online inference over audio-visual streams.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26485","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MMLab CUHK","organizationType":"academic-lab","sourceUrl":"https://github.com/Lucky-Lance/OmniInteract","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception","Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_omnilayout_9c41426f","familyId":"bmf_c7f5b7944f24","name":"OmniLayout","oneLine":"OmniLayout evaluates language models on printed-circuit-board (PCB) layout placement reasoning under geometric, routing, and connectivity constraints, with four tasks including geometric placement, routability-aware placement, electrical functionality, and tool-augmented agentic reasoning.","area":"Multimodal","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Semiconductors"],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.03261","pdf":"https://arxiv.org/pdf/2607.03261","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.03261"},"evidence":{"snippet":"To bridge this gap, we introduce OmniLayout, the first benchmark designed to evaluate LLMs on printed-circuit-board (PCB) layout placement reasoning under real-world geometric, routing, and connectivity constraints.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.03261"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OmniLayout evaluates language models on printed-circuit-board (PCB) layout placement reasoning under geometric, routing, and connectivity constraints, with four tasks including geometric placement, routability-aware placement, electrical functionality, and tool-augmented agentic reasoning.","whyItMatters":"The benchmark addresses the gap in evaluating LLMs for practical electronic design automation (EDA) tasks, specifically constraint-aware geometric reasoning in PCB layout, which is critical for real-world design workflows.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ff1c5ccad8c9f7a650a1da75a0fce42ea1c4a454c126e1b59937a4278db41434"},"motivation":"Recent large language models (LLMs) have demonstrated remarkable progress in 3D spatial reasoning, spatial grounding, and fine-grained geometric understanding.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.03261","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_omnimatbench_4692794c","familyId":"bmf_e2542c6f5f83","name":"OmniMatBench","oneLine":"OmniMatBench assesses multimodal reasoning in materials science across 19 subfields with 3,171 expert-curated QA and calculation problems, spanning four domains from fundamental knowledge to applied materials.","area":"Multimodal","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals"],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.29833","pdf":"https://arxiv.org/pdf/2605.29833","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29833"},"evidence":{"snippet":"To fill this gap, we present OmniMatBench, a human-calibrated multimodal reasoning benchmark for materials science.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29833"},"ranking":{},"description":"OmniMatBench assesses multimodal reasoning in materials science across 19 subfields with 3,171 expert-curated QA and calculation problems, spanning four domains from fundamental knowledge to applied materials.","whyItMatters":"Existing materials benchmarks focus on narrow tasks; OmniMatBench provides a broad reasoning benchmark revealing a substantial gap in current MLLMs, guiding AI assistant development in materials research.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9a8d275f49a4b7bd2cb17b9132e25608cb7d483caf39f5e464bff3ba54d93e26"},"motivation":"As multimodal language models play an increasingly important role in scientific research, materials science offers a critical testbed due to its interdisciplinary, multimodal, and application-driven nature.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29833","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_999b52bdc3ac2d6b","familyId":"catalog_family_999b52bdc3ac2d6b","name":"OmniMath","oneLine":"A Universal Olympiad Level Mathematic Benchmark for Large Language Models containing 4,428 competition-level problems with rigorous human annotation, categorized into over 33 sub-domains and spanning more than 10 distinct difficulty levels","description":"A Universal Olympiad Level Mathematic Benchmark for Large Language Models containing 4,428 competition-level problems with rigorous human annotation, categorized into over 33 sub-domains and spanning more than 10 distinct difficulty levels","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/omnimath","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_999b52bdc3ac2d6b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/omnimath"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"omnimath","url":"https://llm-stats.com/benchmarks/omnimath","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_omnimech_c453faaf","familyId":"bmf_67e7a1510b88","name":"OmniMech","oneLine":"OmniMech evaluates vision-language models on four tasks: CAD program synthesis from engineering drawings, diagram-to-3D reasoning, annotation-grounded reasoning, and tool-augmented agentic reasoning using industrial mechanical data with 251k drawings and associated CAD models.","area":"Science & Engineering","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Manufacturing"],"capabilities":["Geometric reasoning"],"topics":["CAD","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.05539","pdf":"https://arxiv.org/pdf/2608.05539","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.05539"},"evidence":{"snippet":"We introduce OmniMech, the first million-scale benchmark for evaluating VLMs on executable CAD generation from industrial manufacturing data.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.05539"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OmniMech evaluates vision-language models on four tasks: CAD program synthesis from engineering drawings, diagram-to-3D reasoning, annotation-grounded reasoning, and tool-augmented agentic reasoning using industrial mechanical data with 251k drawings and associated CAD models.","whyItMatters":"Existing benchmarks focus on coarse 3D objects; OmniMech addresses the need for evaluating VLMs on fine-grained, dimensioned mechanical designs, providing a standardized testbed to assess progress in executable CAD generation and 3D reconstruction for industrial applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ee1600bbdd2e77b47da49e24a0230a697e8e13c9dc20197380711c11b6c33925"},"motivation":"Recent vision-language models (VLMs) can generate executable CAD programs from images, but existing methods mainly target coarse, general-purpose 3D objects and rarely address the fine-grained geometry and millimeter-level tolerances required in industrial mechanical design.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.05539","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_omniopt_f648d471","familyId":"bmf_15c40ef3c174","name":"OmniOpt","oneLine":"A cross-domain benchmark for comparing optimizers in large-scale model training. It covers 24+ optimizers across two stages: Stage 1 sweeps on C4 with LLaMA-3 architectures (60M to 1B), and Stage 2 transfers to FineWeb-Edu with four architectures (Transformer++, GLA, DeltaNet, Gated DeltaNet) at 340M and 1B scales. Controlled-variable protocol with fixed architecture, data, and schedule settings.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-04","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.04033","pdf":"https://arxiv.org/pdf/2607.04033","project":null,"code":"https://github.com/OpenRaiser/OmniOpt","data":null,"hfPaper":"https://huggingface.co/papers/2607.04033"},"evidence":{"snippet":"We therefore present OmniOpt, a unified survey and benchmark cookbook of optimizers for the research community.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":77,"hfDailySubmittedAt":"2026-07-07T00:00:00.000Z","githubStars":39,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.04033"},"ranking":{"90d":{"score":54,"rank":41,"coverage":0.7,"confidence":"Medium"}},"description":"A cross-domain benchmark for comparing optimizers in large-scale model training. It covers 24+ optimizers across two stages: Stage 1 sweeps on C4 with LLaMA-3 architectures (60M to 1B), and Stage 2 transfers to FineWeb-Edu with four architectures (Transformer++, GLA, DeltaNet, Gated DeltaNet) at 340M and 1B scales. Controlled-variable protocol with fixed architecture, data, and schedule settings.","whyItMatters":"Optimizer selection is a system-level decision impacting compute, memory, and tuning budget. This benchmark provides a unified protocol for comparing methods across multiple scales and architectures, offering reproducible evidence for practitioners to choose optimizers based on measured training objectives and trade-offs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"297971f39f8cba69d19bc299168f8baf176f30baca420701c21f92ebd23f4510"},"motivation":"Optimizer selection for large-scale model training has become a system-level design decision constrained jointly by compute, memory, tuning budget, and task diversity, yet the landscape of over one hundred methods remains fragmented.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.04033","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"OmniOpt Team","organizationType":"community","sourceUrl":"https://github.com/OpenRaiser/OmniOpt","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_omniphys_ae437965","familyId":"bmf_1f087830f505","name":"OmniPhys","oneLine":"Evaluates multimodal physics understanding, reasoning, and generation on 15,246 questions with 19,850 images from middle-school to university levels.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-26","firstSeenAt":"2026-08-27","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25398","pdf":"https://arxiv.org/pdf/2608.25398","project":null,"code":"https://github.com/ECNU-RAIL/OmniPhys-EMNLP2026","data":null,"hfPaper":null},"evidence":{"snippet":"To fill this gap, we introduce OmniPhys, a large-scale benchmark for multimodal physics understanding and reasoning, covering middle school through university-level problems from Chinese Educational Corpora.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25398"},"ranking":{"30d":{"score":23,"rank":110,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":314,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates multimodal physics understanding, reasoning, and generation on 15,246 questions with 19,850 images from middle-school to university levels.","whyItMatters":"Fills the gap in comprehensive physics benchmarks for MLLMs, including structured diagram generation and fine-grained reasoning analysis.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"cfeee56d6d1abc7b9597e93ad0b6d52394fe5dfb509da78d7d73b80444097a94"},"motivation":"Multimodal Large Language Models (MLLMs) have demonstrated strong abilities in solving diverse visual and textual reasoning tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:12:36.511123Z","model":"deepseek-v4-pro","decisionReason":"OmniPhys is named in the abstract and linked to a GitHub repository with data and code, indicating a reusable public benchmark accepted to EMNLP 2026 Findings.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce OmniPhys, a large-scale benchmark for multimodal physics understanding and reasoning"},"publication":{"status":"acceptance_claimed","venue":"Findings of EMNLP 2026","evidence":"Accepted to Findings of EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2608.25398","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-27T04:12:10.575570Z"},"venueAttempts":[{"venueName":"Findings of EMNLP 2026","reviewStatus":"accepted","decisionRaw":"Accepted to Findings of EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.25398","observedAt":"2026-08-27T04:12:10.575570Z","rawValue":"Accepted to Findings of EMNLP 2026","level":"author-claim"}]}],"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"Peer-reviewed venue acceptance and a large multimodal dataset with Chinese educational coverage; domain may limit broader adoption."},"evaluationMode":"public_reusable","publishers":[{"name":"ECNU-RAIL","organizationType":"academic-lab","sourceUrl":"https://github.com/ECNU-RAIL/OmniPhys-EMNLP2026","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_omniphys_eb0844b7","familyId":"bmf_1f087830f505","name":"OmniPhys","oneLine":"OmniPhys is a benchmark of 1,551 text-to-image generation samples grounded in a Physical Knowledge Graph, aligned with PhET simulations and curricula. It evaluates physical commonsense in generated images using a dual-path verification protocol.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25641","pdf":"https://arxiv.org/pdf/2607.25641","project":null,"code":"https://github.com/zjukg/OmniPhys","data":null,"hfPaper":"https://huggingface.co/papers/2607.25641"},"evidence":{"snippet":"To address these challenges, we introduce OmniPhys, a rigorous benchmark of 1,551 samples grounded in a Physical Knowledge Graph.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25641"},"ranking":{"90d":{"score":23,"rank":363,"coverage":0.55,"confidence":"Low"}},"description":"OmniPhys is a benchmark of 1,551 text-to-image generation samples grounded in a Physical Knowledge Graph, aligned with PhET simulations and curricula. It evaluates physical commonsense in generated images using a dual-path verification protocol.","whyItMatters":"Existing benchmarks use coarse descriptions and fail to diagnose specific physical principles. OmniPhys provides a fine-grained, curriculum-aligned evaluation to identify systemic physical reasoning gaps in image generation models, supporting targeted improvement.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"97c19e30b9cdcf993a1194adeb7f5614e69344cc3e4a4a4ee494d460edd5a4c9"},"motivation":"While text-to-image models exhibit remarkable visual fidelity, they frequently violate fundamental physical commonsense.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"KDD 2026 DB track","evidence":"accepted by KDD 2026 DB track","evidenceUrl":"https://arxiv.org/abs/2607.25641","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"KDD 2026 DB track","reviewStatus":"accepted","decisionRaw":"accepted by KDD 2026 DB track","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.25641","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"accepted by KDD 2026 DB track","level":"author-claim"}]}],"publishers":[{"name":"ZJUKG","organizationType":"academic-lab","sourceUrl":"https://github.com/zjukg/OmniPhys","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_omniref-bench_79744dc3","familyId":"bmf_0e055f109041","name":"OmniRef-Bench","oneLine":"OmniRef-Bench is introduced in the paper but primarily serves to evaluate the proposed DyRef method. It is not presented as a standalone public benchmark.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26947","pdf":"https://arxiv.org/pdf/2606.26947","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.26947"},"evidence":{"snippet":"To better assess model performance on complex MRIG tasks, we introduce OmniRef-Bench, a benchmark that covers complex combinations of reference image types and a large number of reference images.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26947"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OmniRef-Bench is introduced in the paper but primarily serves to evaluate the proposed DyRef method. It is not presented as a standalone public benchmark.","whyItMatters":"N/A","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3a3feb743ca8bdc00dc0bacc0e3fe35416491a5318e8eb8f7cc3b3882e1526ca"},"motivation":"While personalized image generation has achieved remarkable progress, multi-reference image generation (MRIG) remains a challenging task.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV2026","evidence":"Accepted by ECCV2026","evidenceUrl":"https://arxiv.org/abs/2606.26947","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV2026","reviewStatus":"accepted","decisionRaw":"Accepted by ECCV2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.26947","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by ECCV2026","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_omnirouting_ca769ccd","familyId":"bmf_f17dee6f17fd","name":"OmniRouting","oneLine":"OmniRouting evaluates LLMs on PCB routing reasoning under real-world constraints with 1,681 industrial designs across four tasks: geometric routing, design-rule-aware routing, electrical functionality, and tool-augmented agentic routing.","area":"Multimodal","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Semiconductors"],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04434","pdf":"https://arxiv.org/pdf/2608.04434","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.04434"},"evidence":{"snippet":"To bridge this gap, we introduce OmniRouting, the first large-scale benchmark designed to evaluate LLMs on printed-circuit-board (PCB) routing reasoning under real-world industrial design-rule, manufacturability, and connectivity constraints.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04434"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OmniRouting evaluates LLMs on PCB routing reasoning under real-world constraints with 1,681 industrial designs across four tasks: geometric routing, design-rule-aware routing, electrical functionality, and tool-augmented agentic routing.","whyItMatters":"Current benchmarks do not cover routing under strict geometric, topological, and electrical constraints. OmniRouting fills this gap, exposing limitations in path planning and rule adherence for LMMs in EDA.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a2fef8a361f3635d8a96ce754e82771c093093fe7dea6a0095571c299168c807"},"motivation":"Recent large language models (LLMs) have demonstrated remarkable progress in constraint-aware navigation, maze reasoning, and graph reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04434","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_23575a8f73f0d1af","familyId":"catalog_family_23575a8f73f0d1af","name":"OmniScience","oneLine":"OmniScience is a broad scientific knowledge and reasoning benchmark that measures both answer accuracy and non-hallucination (calibrated abstention) across science domains.","description":"OmniScience is a broad scientific knowledge and reasoning benchmark that measures both answer accuracy and non-hallucination (calibrated abstention) across science domains.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","Science"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/omniscience","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_23575a8f73f0d1af"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/omniscience"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"omniscience","url":"https://llm-stats.com/benchmarks/omniscience","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","science"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_c26d004be04d59b6","familyId":"catalog_family_c26d004be04d59b6","name":"OmniScience (non-hallucination rate)","oneLine":"OmniScience variant that reports the non-hallucination rate, defined as one minus the hallucination rate.","description":"OmniScience variant that reports the non-hallucination rate, defined as one minus the hallucination rate.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","Science"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/omniscience-non-hallucination-rate","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c26d004be04d59b6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/omniscience-non-hallucination-rate"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"omniscience-non-hallucination-rate","url":"https://llm-stats.com/benchmarks/omniscience-non-hallucination-rate","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","science"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_omnitom_56af1224","familyId":"bmf_e7330cebd753","name":"OmniToM","oneLine":"OmniToM evaluates theory of mind in LLMs by requiring explicit belief modeling, extracting belief propositions and labeling them with seven-dimensional schema labels across 895 stories.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26322","pdf":"https://arxiv.org/pdf/2605.26322","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.26322"},"evidence":{"snippet":"In order to address this research gap, we introduce OmniToM, a benchmark that directly evaluates these representations by requiring explicit modeling of belief structures for all relevant actors within a narrative.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26322"},"ranking":{},"description":"OmniToM evaluates theory of mind in LLMs by requiring explicit belief modeling, extracting belief propositions and labeling them with seven-dimensional schema labels across 895 stories.","whyItMatters":"It addresses the gap of end-point question answering in ToM evaluation by forcing explicit mental-state representation, potentially revealing actor-specific belief-tracking bottlenecks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"87223a18c750df8b2885abbbc7db0603d64c86f3bc7228e21475900154520c8c"},"motivation":"Theory of Mind (ToM), the ability to infer others' knowledge, intentions, and emotions, is commonly evaluated in large language models (LLMs) using end-point question answering, where performance is judged solely by the final answer to a social reasoning query.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26322","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_omnitraffic_61df9afe","familyId":"bmf_8c1a75de50ea","name":"OmniTraffic","oneLine":"OmniTraffic is a controllable generation pipeline and benchmark for spatio-temporal traffic reasoning. It provides 8M VQA samples and a 3K human-verified test set across 12 reconstructed 3D intersections, with a three-level task hierarchy covering scene perception, multi-view and temporal reasoning, and decision support.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15749","pdf":"https://arxiv.org/pdf/2606.15749","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.15749"},"evidence":{"snippet":"We introduce OmniTraffic, a controllable generation pipeline and benchmark for spatio-temporal traffic reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15749"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OmniTraffic is a controllable generation pipeline and benchmark for spatio-temporal traffic reasoning. It provides 8M VQA samples and a 3K human-verified test set across 12 reconstructed 3D intersections, with a three-level task hierarchy covering scene perception, multi-view and temporal reasoning, and decision support.","whyItMatters":"OmniTraffic fills the gap in evaluating structure-aware traffic reasoning under controlled conditions, where existing traffic benchmarks focus on passive recognition. The extensible pipeline allows configurable scenarios for reproducible evaluation and simulation-generated supervision for model fine-tuning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"101a4b0d802cbf08a3ae456d3d0b48bd4f9708849d172bcec6ffb102bc7112ae"},"motivation":"Traffic scene understanding requires models to reason beyond object recognition, including lane topology, multi-view geometry, temporal evolution, and signal-phase semantics.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15749","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_oncotraj_0ff8775a","familyId":"bmf_bdc2d0e2d2ef","name":"OncoTraj","oneLine":"OncoTraj is a public benchmark for predicting acquired resistance to first-line osimertinib in EGFR-mutant non-small-cell lung cancer. It provides a harmonized dataset of 813 patients from three real-world sources, with locked splits, an evaluation harness, and six baselines, defining three tasks: 12-month progression classification, time-to-progression regression, and resistance mechanism classification.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11144","pdf":"https://arxiv.org/pdf/2606.11144","project":null,"code":"https://github.com/span-ai-labs/oncotraj","data":"https://huggingface.co/datasets/span-ai-labs/oncotraj-v1","hfPaper":"https://huggingface.co/papers/2606.11144"},"evidence":{"snippet":"We introduce OncoTraj, a public benchmark of 813 EGFR-mutant NSCLC patients receiving first-line osimertinib, harmonized from three real-world clinical-genomic sources: MSK-CHORD (672 patients), AACR Project GENIE BPC NSCLC (34 patients), and the FLAURA molecular-resistance supplement (107 patients).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":39,"hfDatasetLikes":1},"source":{"type":"arxiv","id":"2606.11144"},"ranking":{"90d":{"score":17,"rank":397,"coverage":0.85,"confidence":"High","datasetDownloadRank":64,"datasetRankPopulation":66}},"description":"OncoTraj is a public benchmark for predicting acquired resistance to first-line osimertinib in EGFR-mutant non-small-cell lung cancer. It provides a harmonized dataset of 813 patients from three real-world sources, with locked splits, an evaluation harness, and six baselines, defining three tasks: 12-month progression classification, time-to-progression regression, and resistance mechanism classification.","whyItMatters":"OncoTraj fills the gap of a standardized, public evaluation for longitudinal resistance prediction, offering reproducible baseline results and leakage-audited splits. It provides a practical means to compare models and highlights the limitation of single-timepoint features, guiding future data collection and algorithm development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"45e7b3d746c923009561febbf9ece9a74aebae17d8fadd6f94d4d054cf38b6ec"},"motivation":"Resistance to first-line osimertinib in EGFR-mutant non-small-cell lung cancer (NSCLC) is the canonical example of predictable clonal evolution under therapeutic pressure, yet no public benchmark exists for training or evaluating computational models on the corresponding longitudinal patient trajectories.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11144","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Span AI Labs","organizationType":"company-research-lab","sourceUrl":"https://github.com/span-ai-labs/oncotraj","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_oncotriad-qa_ae032868","familyId":"bmf_f1efe287a4be","name":"OncoTriad-QA","oneLine":"OncoTriad-QA is a patient-level benchmark for pan-cancer question answering, integrating radiology, pathology, genomics, and clinical metadata from TCGA. It includes 86.1k questions across 9,281 cases and 32 cancer cohorts, with annotations derived from curated labels and diagnostic reports.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02615","pdf":"https://arxiv.org/pdf/2608.02615","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.02615"},"evidence":{"snippet":"We introduce OncoTriad-QA, a patient-level radiology-pathology-genomics benchmark for pan-cancer question answering.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02615"},"ranking":{},"description":"OncoTriad-QA is a patient-level benchmark for pan-cancer question answering, integrating radiology, pathology, genomics, and clinical metadata from TCGA. It includes 86.1k questions across 9,281 cases and 32 cancer cohorts, with annotations derived from curated labels and diagnostic reports.","whyItMatters":"Existing medical LLM benchmarks often focus on isolated modalities. OncoTriad-QA could address the gap in evaluating integrated, patient-level reasoning across multiple evidence types, which is crucial for real-world oncology decision support.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9b349298381d03d3f706f8c6d7ed3c680ea5aa197cead9d8feec58605efcc2be"},"motivation":"Cancer diagnosis and characterization require integrating complementary evidence from radiology, pathology, genomics, and clinical metadata.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02615","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_fca7f5666d5002e2","familyId":"catalog_family_fca7f5666d5002e2","name":"OneMillion Bench","oneLine":"OneMillion Bench evaluates AI agents on high-economic-value tasks that require sustained, reliable execution across long-horizon real-world workflows.","description":"OneMillion Bench evaluates AI agents on high-economic-value tasks that require sustained, reliable execution across long-horizon real-world workflows.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/onemillion-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fca7f5666d5002e2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/onemillion-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"onemillion-bench","url":"https://llm-stats.com/benchmarks/onemillion-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","agents"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Long Context & Memory"],"domainScope":"general"},{"id":"bm_onepot-bench_ffd24571","familyId":"bmf_ddcd7659e7a2","name":"onepot-Bench","oneLine":"onepot-Bench 0 is a proprietary benchmark suite evaluating language models on synthetic chemistry capabilities, including cheminformatics literacy, safety behavior, and reaction outcome prediction.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals"],"capabilities":[],"topics":["cs.LG"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02595","pdf":"https://arxiv.org/pdf/2608.02595","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.02595"},"evidence":{"snippet":"We introduce onepot-Bench 0, a proprietary benchmark suite for evaluating language models on synthetic chemistry capabilities relevant to wet-lab execution.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02595"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"onepot-Bench 0 is a proprietary benchmark suite evaluating language models on synthetic chemistry capabilities, including cheminformatics literacy, safety behavior, and reaction outcome prediction.","whyItMatters":"Targets skills needed for reliable laboratory decisions, addressing limitations of public data benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"693005891e1b35ecdfaae1723d081526d7410b397c4f82f2556615eff7466b72"},"motivation":"Language models are playing an increasingly important role in laboratory science, performing tasks such as experiment planning, execution, and post-hoc analysis.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02595","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_opai-bench_24958874","familyId":"bmf_a2240db81ac2","name":"OpAI-Bench","oneLine":"OpAI-Bench evaluates AI-text detection across document, sentence, token, and span granularities using operation-guided progressive human-to-AI revision trajectories with nine versions per sample.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06481","pdf":"https://arxiv.org/pdf/2606.06481","project":null,"code":"https://github.com/VILA-Lab/OpAI-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.06481"},"evidence":{"snippet":"We introduce OpAI-Bench, an operation-guided benchmark for studying progressive human-to-AI text transformation across document, sentence, token, and span granularities.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":10,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06481"},"ranking":{"90d":{"score":40,"rank":147,"coverage":0.55,"confidence":"Low"}},"description":"OpAI-Bench evaluates AI-text detection across document, sentence, token, and span granularities using operation-guided progressive human-to-AI revision trajectories with nine versions per sample.","whyItMatters":"Existing benchmarks focus on static outputs, missing progressive co-editing. OpAI-Bench provides a controlled testbed with multi-granularity provenance to analyze how AI-authorship signals emerge and accumulate, revealing non-monotonic detection patterns.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"55486acd656d173145f39ba67059a140c6a02151760bcdf38154e0d7b5fc5e47"},"motivation":"As AI writing assistants become increasingly integrated into real-world drafting and revision workflows, many documents are no longer purely human-written or AI-generated, but instead result from progressive human-AI co-editing.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.06481","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"VILA-Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/VILA-Lab/OpAI-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_bcfbbc5925efb33a","familyId":"catalog_family_bcfbbc5925efb33a","name":"Open-rewrite","oneLine":"OpenRewriteEval is a benchmark for evaluating open-ended rewriting of long-form texts, covering a wide variety of rewriting types expressed through natural language instructions including formality, expansion, conciseness, paraphrasing, and tone and style transfer.","description":"OpenRewriteEval is a benchmark for evaluating open-ended rewriting of long-form texts, covering a wide variety of rewriting types expressed through natural language instructions including formality, expansion, conciseness, paraphrasing, and tone and style transfer.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Writing"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/open-rewrite","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bcfbbc5925efb33a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/open-rewrite"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"open-rewrite","url":"https://llm-stats.com/benchmarks/open-rewrite","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","writing"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_775c292833e334b2","familyId":"catalog_family_775c292833e334b2","name":"OpenAI MMLU","oneLine":"MMLU (Massive Multitask Language Understanding) is a comprehensive benchmark that measures a text model's multitask accuracy across 57 diverse academic and professional subjects. The test covers elementary mathematics, US history, computer science, law, morality, business ethics, clinical knowledge, and many other domains spanning STEM, humanities, social sciences, and professional fields. To attain high accuracy, models must possess extensive world knowledge and problem-solving ability.","description":"MMLU (Massive Multitask Language Understanding) is a comprehensive benchmark that measures a text model's multitask accuracy across 57 diverse academic and professional subjects. The test covers elementary mathematics, US history, computer science, law, morality, business ethics, clinical knowledge, and many other domains spanning STEM, humanities, social sciences, and professional fields. To attain high accuracy, models must possess extensive world knowledge and problem-solving ability.","area":"Mathematical Reasoning","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Legal","Math","Physics","Psychology","Reasoning","Finance","General","Healthcare","Chemistry","Economics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/openai-mmlu","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_775c292833e334b2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/openai-mmlu"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"openai-mmlu","url":"https://llm-stats.com/benchmarks/openai-mmlu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["legal","math","physics","psychology","reasoning","finance","general","healthcare","chemistry","economics"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"catalog_b1a7569e9666f49d","familyId":"catalog_family_b1a7569e9666f49d","name":"OpenAI-MRCR: 2 needle 128k","oneLine":"Multi-round Co-reference Resolution (MRCR) benchmark for evaluating an LLM's ability to distinguish between multiple needles hidden in long context. Models are given a long, multi-turn synthetic conversation and must retrieve a specific instance of a repeated request, requiring reasoning and disambiguation skills beyond simple retrieval.","description":"Multi-round Co-reference Resolution (MRCR) benchmark for evaluating an LLM's ability to distinguish between multiple needles hidden in long context. Models are given a long, multi-turn synthetic conversation and must retrieve a specific instance of a repeated request, requiring reasoning and disambiguation skills beyond simple retrieval.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/openai-mrcr:-2-needle-128k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b1a7569e9666f49d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/openai-mrcr:-2-needle-128k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"openai-mrcr:-2-needle-128k","url":"https://llm-stats.com/benchmarks/openai-mrcr:-2-needle-128k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":9,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_50454980643700d2","familyId":"catalog_family_50454980643700d2","name":"OpenAI-MRCR: 2 needle 1M","oneLine":"Multi-Round Co-reference Resolution benchmark that tests an LLM's ability to distinguish between multiple similar needles hidden in long conversations. Models must reproduce specific instances of content (e.g., 'Return the 2nd poem about tapirs') from multi-turn synthetic conversations, requiring reasoning about context, ordering, and subtle differences between similar outputs.","description":"Multi-Round Co-reference Resolution benchmark that tests an LLM's ability to distinguish between multiple similar needles hidden in long conversations. Models must reproduce specific instances of content (e.g., 'Return the 2nd poem about tapirs') from multi-turn synthetic conversations, requiring reasoning about context, ordering, and subtle differences between similar outputs.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/openai-mrcr:-2-needle-1m","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_50454980643700d2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/openai-mrcr:-2-needle-1m"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"openai-mrcr:-2-needle-1m","url":"https://llm-stats.com/benchmarks/openai-mrcr:-2-needle-1m","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_b466792b10a39034","familyId":"catalog_family_b466792b10a39034","name":"OpenAI-MRCR: 2 needle 256k","oneLine":"Multi-Round Co-reference Resolution (MRCR) benchmark that tests long-context reasoning by evaluating a model's ability to distinguish between similar outputs, reason about ordering, and reproduce specific content from multi-turn conversations containing multiple writing requests on overlapping topics at 256k tokens.","description":"Multi-Round Co-reference Resolution (MRCR) benchmark that tests long-context reasoning by evaluating a model's ability to distinguish between similar outputs, reason about ordering, and reproduce specific content from multi-turn conversations containing multiple writing requests on overlapping topics at 256k tokens.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/openai-mrcr:-2-needle-256k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b466792b10a39034"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/openai-mrcr:-2-needle-256k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"openai-mrcr:-2-needle-256k","url":"https://llm-stats.com/benchmarks/openai-mrcr:-2-needle-256k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_openbenchmark_199b1148","familyId":"bmf_9da9d25e3255","name":"OpenBenchmark","oneLine":"Evaluates AI agents and models on user-defined tasks generated from imported agent trajectories, scoring correctness, cost, and speed.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/crispyberry/OpenBenchmark","pdf":null,"project":"https://pi.dev","code":"https://github.com/crispyberry/OpenBenchmark","data":null,"hfPaper":null},"evidence":{"snippet":"ai-agents benchmark evaluation llm llm-evaluation nextjs observability openrouter postgres typescript # OpenBenchmark **English** · [简体中文](README.zh-CN.md) · [日本語](README.ja.md) · [Français](README.fr.md) · [Español](README.es.md) **Turn real agent trajectories into runnable benchmarks, then rank models, harnesses, and your own agents on score, cost, and speed.** Standard benchmarks tell you which model is good at someone else's problem.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:crispyberry/openbenchmark"},"ranking":{"30d":{"score":28,"rank":90,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":260,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates AI agents and models on user-defined tasks generated from imported agent trajectories, scoring correctness, cost, and speed.","whyItMatters":"Enables organizations to build benchmarks from their own agent logs, producing rankings that reflect real operational workloads rather than generic tasks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"9021bec27711d896be6d7d85022ab763e4836dbca4de291a4ec7ee56e125a0cc"},"motivation":"OpenBenchmark Turn real agent trajectories into runnable benchmarks, then rank models, harnesses, and your own agents on score, cost, and speed.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The repository provides extensive documentation and code, but no independent paper, OpenReview entry, or official benchmark site is supplied; only a GitHub README is available, so the formal release status is unclear."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/crispyberry/OpenBenchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"attentionForecast":{"score":30,"confidence":"Low","horizon":"7d","reason":"A niche developer tool with moderate discovery potential from GitHub search and the multilingual README, but no academic paper or community evidence is provided to drive initial attention."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_113bcabbd0e2059f","familyId":"catalog_family_113bcabbd0e2059f","name":"OpenBookQA","oneLine":"OpenBookQA is a question-answering dataset modeled after open book exams for assessing human understanding. It contains 5,957 multiple-choice elementary-level science questions that probe understanding of 1,326 core science facts and their application to novel situations, requiring combination of open book facts with broad common knowledge through multi-hop reasoning.","description":"OpenBookQA is a question-answering dataset modeled after open book exams for assessing human understanding. It contains 5,957 multiple-choice elementary-level science questions that probe understanding of 1,326 core science facts and their application to novel situations, requiring combination of open book facts with broad common knowledge through multi-hop reasoning.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/1809.02789","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_113bcabbd0e2059f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/openbookqa"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/openbookqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"openBookQa","url":"https://benchlm.ai/benchmarks/openbookqa","paperUrl":"https://arxiv.org/abs/1809.02789","year":"2018","fullName":"OpenBookQA","format":"4-way multiple choice","tasks":"Elementary science questions","successorKey":null},{"catalog":"llm-stats","sourceId":"openbookqa","url":"https://llm-stats.com/benchmarks/openbookqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","general"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_openhaldet_b75bc310","familyId":"bmf_34d27080ff9a","name":"OpenHalDet","oneLine":"OpenHalDet is a unified benchmark for hallucination detection across 17 datasets, supporting black-box, gray-box, and white-box detectors with standardized pipelines, scoring via AUROC and Cost@N.","area":"Vision & 3D","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06959","pdf":"https://arxiv.org/pdf/2606.06959","project":null,"code":"https://github.com/Nellie179/Hallucination-Detection","data":null,"hfPaper":"https://huggingface.co/papers/2606.06959"},"evidence":{"snippet":"We introduce OpenHalDet, a unified benchmark for hallucination detection across diverse generation scenarios.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06959"},"ranking":{"90d":{"score":17,"rank":396,"coverage":0.7,"confidence":"Medium"}},"description":"OpenHalDet is a unified benchmark for hallucination detection across 17 datasets, supporting black-box, gray-box, and white-box detectors with standardized pipelines, scoring via AUROC and Cost@N.","whyItMatters":"It standardizes hallucination detection evaluation, enabling fair comparison across diverse methods and providing a systematic view of detector performance in LLM applications, addressing inconsistencies and limited coverage.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"57feff3fcd46557bb0cf3612ee1ccaff2fa046913059bc2ed881172f376ab3a5"},"motivation":"Hallucination detection is essential for the reliable deployment of large language models (LLMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.06959","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_06190eb7d3e975c6","familyId":"catalog_family_06190eb7d3e975c6","name":"OpenHands Index","oneLine":"A holistic coding-agent benchmark that evaluates AI agents across issue resolution, frontend work, greenfield development, testing, and information gathering.","description":"A holistic coding-agent benchmark that evaluates AI agents across issue resolution, frontend work, greenfield development, testing, and information gathering.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://index.openhands.dev/about","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_06190eb7d3e975c6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/openhandsindex"}],"catalogSources":[{"catalog":"benchlm","sourceId":"openHandsIndex","url":"https://benchlm.ai/benchmarks/openhandsindex","paperUrl":"https://index.openhands.dev/about","year":"2025","fullName":"OpenHands Index","format":"Macro-average across five coding-agent categories","tasks":"SWE-bench Verified, SWE-bench Multimodal, Commit0, SWT-bench Verified, and GAIA","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_openharmony-bench_a2c42703","familyId":"bmf_ad07fa1e0a63","name":"OpenHarmony Bench","oneLine":"OpenHarmony Bench evaluates coding agents on 153 app-level ArkTS tasks across three input sources, with 242 feature points and device-based verification.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.16022","pdf":"https://arxiv.org/pdf/2608.16022","project":"https://bench.matrix.openharmony.cn/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.16022"},"evidence":{"snippet":"We present OPENHARMONY BENCH, an app-level coding benchmark for evaluating LLM-based coding agents on OpenHarmony ArkTS applications.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.16022"},"ranking":{"30d":{"score":41,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":45,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"OpenHarmony Bench evaluates coding agents on 153 app-level ArkTS tasks across three input sources, with 242 feature points and device-based verification.","whyItMatters":"It measures end-to-end app-level correctness, filling the gap between function-level coding benchmarks and real-world app development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a954e61b07650b8569ccbd8e7f1dc95b0b4603681721a0ca94040c81faf4537a"},"motivation":"We present OPENHARMONY BENCH, an app-level coding benchmark for evaluating LLM-based coding agents on OpenHarmony ArkTS applications.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.16022","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"OpenHarmony Community","organizationType":"community","sourceUrl":"https://bench.matrix.openharmony.cn/","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"openHarmonyBench","url":"https://benchlm.ai/benchmarks/openharmonybench","paperUrl":"https://arxiv.org/abs/2608.16022","year":"2026","fullName":"OpenHarmony Bench v1.0","format":"Task completion through DevEco Code","tasks":"153 app-development and bug-fix tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0},{"id":"catalog_8055bd7cf670b6e1","familyId":"catalog_family_8055bd7cf670b6e1","name":"OpenRCA","oneLine":"OpenRCA is a benchmark for evaluating AI models on root cause analysis tasks. For each failure case, the model receives 1 point if all generated root-cause elements match the ground-truth ones, and 0 points if any mismatch is identified. The overall accuracy is the average score across all failure cases.","description":"OpenRCA is a benchmark for evaluating AI models on root cause analysis tasks. For each failure case, the model receives 1 point if all generated root-cause elements match the ground-truth ones, and 0 points if any mismatch is identified. The overall accuracy is the average score across all failure cases.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/openrca","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8055bd7cf670b6e1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/openrca"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"openrca","url":"https://llm-stats.com/benchmarks/openrca","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_openrtag_074a3702","familyId":"bmf_3b58e1fac77d","name":"OpenRTAG","oneLine":"OpenRTAG evaluates text-attributed graph learning under nine degradation scenarios (sparsity, noise, imbalance) across nine TAG datasets and three downstream tasks. It provides a standardized testbed for robustness comparison.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Robustness"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19108","pdf":"https://arxiv.org/pdf/2607.19108","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19108"},"evidence":{"snippet":"To address this gap, we present OpenRTAG, a robustness benchmark for text-attributed graph learning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19108"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OpenRTAG evaluates text-attributed graph learning under nine degradation scenarios (sparsity, noise, imbalance) across nine TAG datasets and three downstream tasks. It provides a standardized testbed for robustness comparison.","whyItMatters":"Real-world TAGs suffer from data quality issues, but evidence on robustness is fragmented. OpenRTAG unifies these scenarios, enabling systematic evaluation and comparison of mitigation strategies across model families.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7aa7deeedbb15365d7e46d9f71344777e76b0a4480b78b7c454c16dab3719b1b"},"motivation":"Text-attributed graphs (TAGs) are an important graph data form that combine relational structure with rich node text.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19108","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_opensafeintent_d8b14d02","familyId":"bmf_01b35a219e13","name":"OpenSafeIntent","oneLine":"OpenSafeIntent is a benchmark of controlled prompt-sets that vary user intent while holding the underlying task fixed. Each datapoint contains benign, dual-use, and malicious variants of the same task, and models are evaluated on whether they calibrate assistance across intent shifts rather than only appearing safe on average.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.02047","pdf":"https://arxiv.org/pdf/2607.02047","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.02047"},"evidence":{"snippet":"We introduce OpenSafeIntent, a benchmark of controlled prompt-sets that vary intent while holding the underlying task fixed.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02047"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"OpenSafeIntent is a benchmark of controlled prompt-sets that vary user intent while holding the underlying task fixed. Each datapoint contains benign, dual-use, and malicious variants of the same task, and models are evaluated on whether they calibrate assistance across intent shifts rather than only appearing safe on average.","whyItMatters":"It evaluates safety as intent-calibrated behavior over controlled task variants, addressing limitations of evaluating safety on isolated prompts and providing a way to assess safe completion across subtle intent changes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"59e9506c4747617474f5c97a65f0446bad82e0dc9bb86c1a355d836adf0f3052"},"motivation":"Safe completion requires models to provide useful assistance without enabling harm, but this behavior is difficult to evaluate with isolated prompts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02047","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_openscitoolbench_fd41e9ac","familyId":"bmf_867272727e97","name":"OpenSciToolBench","oneLine":"OpenSciToolBench is a benchmark with 900 tasks across four difficulty levels for evaluating LLM agents in open-world scientific tool acquisition.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.28692","pdf":"https://arxiv.org/pdf/2607.28692","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.28692"},"evidence":{"snippet":"Moreover, we introduce OpenSciToolBench, a benchmark containing 900 realistic tasks across four difficulty levels.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.28692"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OpenSciToolBench is a benchmark with 900 tasks across four difficulty levels for evaluating LLM agents in open-world scientific tool acquisition.","whyItMatters":"The benchmark supports a specific agent system's evaluation and lacks a standalone comparison path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ba5b334155f3e56e65c4ac59f900cbe447132860cd3112ae89e524b6c5a29cc7"},"motivation":"Large language model (LLM) agents have been increasingly adopted in scientific research for organizing and invoking specialized computational tools.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.28692","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_openskillrisk_99823415","familyId":"bmf_2b6a63f1b968","name":"OpenSkillRisk","oneLine":"OpenSkillRisk evaluates LLM-based agents on their ability to recognize and avoid safety risks when using third-party skills, with 263 risky skills in seven threat categories and sandboxed execution.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.20121","pdf":"https://arxiv.org/pdf/2607.20121","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20121"},"evidence":{"snippet":"To support quantitative and qualitative evaluation, we construct OpenSkillRisk, a dedicated safety benchmark containing 263 risky skills collected from public skill marketplaces.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20121"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"OpenSkillRisk evaluates LLM-based agents on their ability to recognize and avoid safety risks when using third-party skills, with 263 risky skills in seven threat categories and sandboxed execution.","whyItMatters":"Third-party skills can introduce latent security vulnerabilities. A dedicated safety benchmark helps assess agent risk reasoning and execution control in open-world scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"791382c5c098f6c0ad65124b0b4e064910f17fc1d230a5537d62d8fe3f2f2a30"},"motivation":"LLM-based agents leverage third-party skills to extend their capabilities in open-world scenarios.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20121","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_opti-agent-bench_f017ef81","familyId":"bmf_4e5dce3ea76c","name":"Opti-Agent-Bench","oneLine":"Opti-Agent-Bench evaluates LLM agents on the end-to-end optimization R&D pipeline, from business-language description to mathematical modeling, algorithm selection, code implementation, and report generation, with modular evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.10768","pdf":"https://arxiv.org/pdf/2607.10768","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.10768"},"evidence":{"snippet":"We introduce Opti-Agent-Bench, an end-to-end benchmark that evaluates Large Language Models (LLMs) across the complete optimization R&D pipeline, from understanding business-language descriptions through mathematical modeling, algorithm selection, and code implementation, to solution report generation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.10768"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Opti-Agent-Bench evaluates LLM agents on the end-to-end optimization R&D pipeline, from business-language description to mathematical modeling, algorithm selection, code implementation, and report generation, with modular evaluation.","whyItMatters":"Optimization benchmarks typically test pre-structured formulations. Opti-Agent-Bench assesses the full pipeline, exposing failure modes like constraint omission and model-code inconsistency invisible under single-metric evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ac89ccae89caa72d72bef818a582d6b749680d08e1bbb7869dc78f93157d3bc9"},"motivation":"LLM-based agents are increasingly deployed to solve optimization problems, yet existing benchmarks evaluate them on pre-structured mathematical formulations that bypass the most critical challenge: translating complex business requirements into correct models and solve efficiently.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.10768","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_38e4916adcf974c5","familyId":"catalog_family_38e4916adcf974c5","name":"OptimBench","oneLine":"Catalog-listed benchmark; original-source verification is pending.","description":"Catalog-listed benchmark; original-source verification is pending.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":[],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/community:64d67847-06bd-423a-923c-c2acfab82281","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_38e4916adcf974c5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/community:64d67847-06bd-423a-923c-c2acfab82281"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"community:64d67847-06bd-423a-923c-c2acfab82281","url":"https://llm-stats.com/benchmarks/community:64d67847-06bd-423a-923c-c2acfab82281","datasetSlug":"optimbench","versionCount":9,"subsetCount":1,"rowCount":null,"updatedAt":"2026-03-06T03:50:56.354179+00:00","community":true}],"catalogCategories":[],"catalogModelCount":12,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_or-space_6e200388","familyId":"bmf_3a4d8160cbad","name":"OR-Space","oneLine":"A full-lifecycle workspace benchmark for industrial optimization agents, evaluating model construction, revision, and grounded explanation across three task modes (Build, Revise, Explain) using executable multi-file workspaces with task-specific evaluators.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.28158","pdf":"https://arxiv.org/pdf/2605.28158","project":null,"code":"https://github.com/0xzhouchenyu/OR-Space","data":null,"hfPaper":"https://huggingface.co/papers/2605.28158"},"evidence":{"snippet":"We introduce OR-Space, a full-lifecycle workspace benchmark for evaluating industrial optimization agents across model construction, model revision, and grounded explanation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":6,"hfDailySubmittedAt":null,"githubStars":12,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28158"},"ranking":{},"description":"A full-lifecycle workspace benchmark for industrial optimization agents, evaluating model construction, revision, and grounded explanation across three task modes (Build, Revise, Explain) using executable multi-file workspaces with task-specific evaluators.","whyItMatters":"Fills the gap in benchmarking LLM agents for real industrial OR workflows, where persistent multi-artifact workspaces and multi-stage lifecycles are central, offering a more realistic evaluation of practical readiness beyond single-shot formulation tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2275bbeddc868adc4d82c8d69ac82a5139372958660412e16d8e3ccf0ac8f78d"},"motivation":"Large language model (LLM) agents are increasingly used to assist with operations research (OR) modeling, yet existing OR-oriented benchmarks often reduce evaluation to one-shot translation from a self-contained problem statement into a mathematical formulation or solver program.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28158","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"OR-Space team","organizationType":"academic-lab","sourceUrl":"https://github.com/0xzhouchenyu/OR-Space","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_oraclephys_04a2c150","familyId":"bmf_467dce972bf6","name":"OraclePhys","oneLine":"Evaluates LLM ranking of stories by inter-story drift and identification of the governing story for multi-story 2-D steel frames, with scoring by top-1 accuracy and Spearman correlation against oracle truth.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-23","firstSeenAt":"2026-08-24","recognitionConfidence":0.4,"links":{"report":"https://github.com/cl1110goo-afk/oraclephys","pdf":null,"project":"https://arxiv.org/abs/2608.17162","code":"https://github.com/cl1110goo-afk/oraclephys","data":null,"hfPaper":null},"evidence":{"snippet":"oraclephys Benchmark, datasets and code for arXiv:2608.17162 benchmark finetuning llm reinforcement-learning structural-engineering # OraclePhys An end-to-end instrument — **benchmark, dataset, training pipeline** — for studying what fine-tuning objectives install in LLMs.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:cl1110goo-afk/oraclephys"},"ranking":{"30d":{"score":48,"rank":73,"coverage":0.55,"confidence":"Low"},"90d":{"score":39,"rank":266,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates LLM ranking of stories by inter-story drift and identification of the governing story for multi-story 2-D steel frames, with scoring by top-1 accuracy and Spearman correlation against oracle truth.","whyItMatters":"Provides exact finite-element ground truth for structural mechanics, enabling controlled comparison of fine-tuning objectives without human labels or LLM judging.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"e17bd57cc6a73e934d5b3b84df3ce4d751ebec1c65bcdaeb14d5bf7987afca33"},"motivation":"oraclephys Benchmark, datasets and code for arXiv:2608.17162 benchmark finetuning llm reinforcement-learning structural-engineering # OraclePhys An end-to-end instrument — **benchmark, dataset, training pipeline** — for studying what fine-tuning objectives install in LLMs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The GitHub repository provides a self-contained static benchmark with JSON prompts and oracle truth, evaluation code, and documented metrics.","canonicalNameSource":"official_readme","canonicalNameEvidence":"# OraclePhys An end-to-end instrument — **benchmark, dataset, training pipeline**"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/cl1110goo-afk/oraclephys","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-24T07:04:32.475961Z"},"attentionForecast":{"score":42,"confidence":"Low","horizon":"7d","reason":"Niche structural mechanics domain limits broad interest, but a public GitHub release with evaluation code and an arXiv paper supports moderate attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_oragentbench_12c4a136","familyId":"bmf_617a34730963","name":"ORAgentBench","oneLine":"ORAgentBench evaluates autonomous agents on end-to-end operations research tasks. Each task involves a natural-language brief, multi-file data, configuration artifacts, and a required submission schema. Agents write and run solution code, validated for schema, hard constraints, and objective quality.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19787","pdf":"https://arxiv.org/pdf/2606.19787","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.19787"},"evidence":{"snippet":"In this work, we introduce ORAgentBench, an execution-grounded benchmark for evaluating autonomous agents on challenging end-to-end operations research tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19787"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"ORAgentBench evaluates autonomous agents on end-to-end operations research tasks. Each task involves a natural-language brief, multi-file data, configuration artifacts, and a required submission schema. Agents write and run solution code, validated for schema, hard constraints, and objective quality.","whyItMatters":"Existing OR evaluations often decouple modeling from solving and rarely test full workflows from artifacts to validated decisions. ORAgentBench provides a realistic, execution-grounded protocol for measuring practical decision-making in operations research, helping to identify strategic weaknesses in current models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0669a1b6c4df86f9221fe089a6208124c0ce3b53ab803c950ffda48a329083c4"},"motivation":"Large language models are increasingly deployed as autonomous agents for multi-step tasks in executable environments, yet their ability to perform realistic operations research (OR) work remains unclear.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19787","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_orca-bench_79a9e07b","familyId":"bmf_ee99b02dd1db","name":"ORCA-bench","oneLine":"ORCA-bench evaluates language model agents on root cause analysis in a production-fidelity oncall setting, using a live OpenTelemetry-instrumented microservice system with 1,079 tasks varying in report specificity, time-to-detection, and fault co-occurrence. Agents access metrics, logs, traces, and source code via real telemetry interfaces.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.28545","pdf":"https://arxiv.org/pdf/2607.28545","project":"https://hub.harborframework.com/datasets/orca-bench/orca-bench","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.28545"},"evidence":{"snippet":"We introduce ORCA-bench, a benchmark that puts general-purpose coding agents in a production-fidelity oncall setting.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.28545"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ORCA-bench evaluates language model agents on root cause analysis in a production-fidelity oncall setting, using a live OpenTelemetry-instrumented microservice system with 1,079 tasks varying in report specificity, time-to-detection, and fault co-occurrence. Agents access metrics, logs, traces, and source code via real telemetry interfaces.","whyItMatters":"Standard coding benchmarks do not capture the complexity of oncall RCA, where agents must reason over noisy, heterogeneous data. ORCA-bench provides a reproducible testbed to measure agent readiness for production reliability tasks, revealing a significant performance gap even for frontier models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cd8c69fe2003f3b637d6ae1908529e54338a40d5249f2b3e299e0934d6c49976"},"motivation":"Large language models can write, patch, and search code, but oncall root cause analysis (RCA) demands something different: reasoning over noisy metrics, logs, traces, and source code, starting from ambiguous user-facing reports, often hours after the incident began.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.28545","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Harbor Framework","organizationType":"benchmark-organization","sourceUrl":"https://hub.harborframework.com/datasets/orca-bench/orca-bench","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_orchbench_2f9fbf5c","familyId":"bmf_11407a7c2e35","name":"OrchBench","oneLine":"OrchBench evaluates multi-agent orchestration plans in isolation using deterministic simulation. It constructs DAGs from real-world tasks and scores plans on result quality, makespan, and token cost without executing worker agents.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25656","pdf":"https://arxiv.org/pdf/2607.25656","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.25656"},"evidence":{"snippet":"We present OrchBench, a simulation-based benchmark for evaluating multi-agent orchestration plans in isolation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25656"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OrchBench evaluates multi-agent orchestration plans in isolation using deterministic simulation. It constructs DAGs from real-world tasks and scores plans on result quality, makespan, and token cost without executing worker agents.","whyItMatters":"OrchBench provides a fast, token-efficient evaluation of orchestration plans, decoupling planning quality from worker capabilities and environmental noise. Its simulated scores correlate strongly with real executions, enabling cost-effective comparison and diagnosis of multi-agent planners.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b87ecaf5ac12614286895adf6b43781cf1be0be29071f9ce08b2f4a9704ae3db"},"motivation":"Complex tasks often decompose into parallelizable yet interdependent subtasks, making orchestration critical to the performance of multi-agent systems (MAS).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25656","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_85bc5f5f054cb9f5","familyId":"catalog_family_85bc5f5f054cb9f5","name":"Organic chemistry V2","oneLine":"Chemistry tasks covering spectroscopy, synthesis planning, reaction prediction, and chemical structure images.","description":"Chemistry tasks covering spectroscopy, synthesis planning, reaction prediction, and chemical structure images.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_85bc5f5f054cb9f5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/organicchemistryv2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"organicChemistryV2","url":"https://benchlm.ai/benchmarks/organicchemistryv2","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Anthropic Organic Chemistry V2 evaluation","format":"Task score","tasks":"Organic chemistry reasoning tasks","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_osguard_1e29e1e9","familyId":"bmf_e3e2aee3f6ec","name":"OSGuard","oneLine":"OSGuard is a dual-granularity benchmark suite for evaluating safety in computer-use agents, with an action-level benchmark for local guardrail decisions and a risk-augmented execution suite for end-to-end evaluation.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15034","pdf":"https://arxiv.org/pdf/2606.15034","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.15034"},"evidence":{"snippet":"We introduce OSGuard, a dual-granularity benchmark suite for evaluating safety in computer-use agents under benign, unchanged user instructions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15034"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OSGuard is a dual-granularity benchmark suite for evaluating safety in computer-use agents, with an action-level benchmark for local guardrail decisions and a risk-augmented execution suite for end-to-end evaluation.","whyItMatters":"Task success alone misses unsafe shortcuts in computer-use agents. OSGuard's dual-granularity design distinguishes local recognition of unsafe actions from full-task safety, providing a more precise diagnosis of guardrail capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2e49568fa636d1d7c39023ab6f3edee2ec0ee1461aebb46c5b08980ba8e3c718"},"motivation":"Computer-use agents are increasingly evaluated by whether they complete realistic desktop and web tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15034","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"lib_osworld","familyId":"family_osworld","name":"OSWorld","oneLine":"Established benchmark family · Computer Use.","area":"Computer Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Computer Use"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2404.07972","pdf":null,"project":"https://os-world.github.io/","code":"https://github.com/xlang-ai/OSWorld","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_osworld"},"ranking":{},"recordType":"family","aliases":[],"sourceAttribution":[{"role":"official-project","url":"https://os-world.github.io/"}],"adoptionRefs":["openai-gpt5","anthropic-claude4","google-gemini25"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Computer Use"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"osWorld","url":"https://benchlm.ai/benchmarks/osworld","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"OSWorld","format":"Interactive GUI evaluation","tasks":"Computer-use tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"osworld","url":"https://llm-stats.com/benchmarks/osworld","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","multimodal","general","agents","vision"],"catalogModelCount":20,"catalogStarCount":0},{"id":"bm_osworld-2-0_ee079e02","familyId":"bmf_f1ccc6d9d057","name":"OSWorld 2.0","oneLine":"Evaluates computer-use agents on 108 long-horizon real-world workflows across everyday and professional tasks, scored by binary completion at 500 steps and partial scores.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29537","pdf":"https://arxiv.org/pdf/2606.29537","project":null,"code":"https://github.com/xlang-ai/OSWorld-V2","data":null,"hfPaper":"https://huggingface.co/papers/2606.29537"},"evidence":{"snippet":"We introduce OSWorld 2.0, a benchmark of 108 long-horizon computer-use workflows across everyday and professional tasks, designed to capture complex and challenging real-world phenomena.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":24,"hfDailySubmittedAt":"2026-06-30T00:00:00.000Z","githubStars":264,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29537"},"ranking":{"90d":{"score":65,"rank":11,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates computer-use agents on 108 long-horizon real-world workflows across everyday and professional tasks, scored by binary completion at 500 steps and partial scores.","whyItMatters":"Captures long-horizon, dynamic, and hidden-state challenges absent in prior benchmarks, revealing that agents fail on constraint tracking and mid-task information.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6a46fc7ce1c8ae35070e7444ec1e6176ab91d714e98416ae04dc584ef408e356"},"motivation":"Existing computer-use benchmarks fail to capture the realism, complexity, and long-horizon demands of real-world computer use, limiting their ability to reveal the limitations of frontier agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.29537","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"XLang Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/xlang-ai/OSWorld-V2","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"osWorld2","url":"https://benchlm.ai/benchmarks/osworld2","paperUrl":"https://arxiv.org/abs/2606.29537","year":"2026","fullName":"OSWorld 2.0","format":"Interactive computer-use evaluation","tasks":"108 long-horizon computer-use workflows","successorKey":null},{"catalog":"llm-stats","sourceId":"osworld-2.0","url":"https://llm-stats.com/benchmarks/osworld-2.0","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","multimodal","general","agents","vision"],"catalogModelCount":6,"catalogStarCount":0},{"id":"catalog_6ae987024f67025b","familyId":"catalog_family_6ae987024f67025b","name":"OSWorld Extended","oneLine":"OSWorld is a scalable, real computer environment benchmark for evaluating multimodal agents on open-ended tasks across Ubuntu, Windows, and macOS. It comprises 369 computer tasks involving real web and desktop applications, OS file I/O, and multi-application workflows. The benchmark evaluates agents' ability to interact with computer interfaces using screenshots and actions in realistic computing environments.","description":"OSWorld is a scalable, real computer environment benchmark for evaluating multimodal agents on open-ended tasks across Ubuntu, Windows, and macOS. It comprises 369 computer tasks involving real web and desktop applications, OS file I/O, and multi-application workflows. The benchmark evaluates agents' ability to interact with computer interfaces using screenshots and actions in realistic computing environments.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/osworld-extended","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6ae987024f67025b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/osworld-extended"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"osworld-extended","url":"https://llm-stats.com/benchmarks/osworld-extended","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","general","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_29abe33c53bcfbfc","familyId":"catalog_family_29abe33c53bcfbfc","name":"OSWorld Screenshot-only","oneLine":"OSWorld Screenshot-only: A variant of the OSWorld benchmark that evaluates multimodal AI agents using only screenshot observations to complete open-ended computer tasks across real operating systems (Ubuntu, Windows, macOS). Tests agents' ability to perform complex workflows involving web apps, desktop applications, file I/O, and multi-application tasks through visual interface understanding and GUI grounding.","description":"OSWorld Screenshot-only: A variant of the OSWorld benchmark that evaluates multimodal AI agents using only screenshot observations to complete open-ended computer tasks across real operating systems (Ubuntu, Windows, macOS). Tests agents' ability to perform complex workflows involving web apps, desktop applications, file I/O, and multi-application tasks through visual interface understanding and GUI grounding.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","General","Grounding","Agents","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/osworld-screenshot-only","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_29abe33c53bcfbfc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/osworld-screenshot-only"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"osworld-screenshot-only","url":"https://llm-stats.com/benchmarks/osworld-screenshot-only","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","general","grounding","agents","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_fa3744e8e53dfc9f","familyId":"catalog_family_fa3744e8e53dfc9f","name":"OSWorld-G","oneLine":"OSWorld-G (Grounding) evaluates screenshot grounding accuracy for OS automation tasks.","description":"OSWorld-G (Grounding) evaluates screenshot grounding accuracy for OS automation tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Grounding","Agents","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/osworld-g","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fa3744e8e53dfc9f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/osworld-g"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"osworld-g","url":"https://llm-stats.com/benchmarks/osworld-g","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","grounding","agents","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"lib_osworld_verified","familyId":"family_osworld","name":"OSWorld-Verified","oneLine":"Established benchmark variant · Computer Use.","area":"Computer Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Computer Use"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2025-01-01","releaseDatePrecision":"year","firstRelease":{"year":2025,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://os-world.github.io/","pdf":null,"project":"https://os-world.github.io/","code":"https://github.com/xlang-ai/OSWorld","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_osworld_verified"},"ranking":{},"recordType":"variant","aliases":["OSWorld Verified"],"sourceAttribution":[{"role":"official-project","url":"https://os-world.github.io/"}],"adoptionRefs":["openai-gpt5"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"}],"catalogDiscoveryRefs":[],"catalogDiscoverySources":[],"usageObservations":[],"variantOf":"lib_osworld","capabilityGroups":["Computer Use"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"osWorldVerified","url":"https://benchlm.ai/benchmarks/osworld-verified","paperUrl":"https://os-world.github.io/","year":"2025","fullName":"OSWorld-Verified","format":"Execution-based interactive task success","tasks":"369 real-world computer tasks (361 when eight Google Drive tasks are excluded)","successorKey":null},{"catalog":"llm-stats","sourceId":"osworld-verified","url":"https://llm-stats.com/benchmarks/osworld-verified","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","multimodal","general","agents","vision"],"catalogModelCount":24,"catalogStarCount":0},{"id":"bm_ov3d-bench_dff2d230","familyId":"bmf_dff0e2aa5738","name":"OV3D-Bench","oneLine":"OV3D-Bench is a diagnostic benchmark for open-vocabulary monocular 3D detection, evaluating localization, semantic robustness, and cross-domain transfer across seven datasets.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.17110","pdf":"https://arxiv.org/pdf/2608.17110","project":null,"code":"https://github.com/mgladkova/ov3d-bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.17110"},"evidence":{"snippet":"To address this, we introduce OV3D-Bench, a diagnostic benchmark that compares open-vocabulary monocular 3D detectors under deployment-realistic conditions across seven indoor and outdoor datasets.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17110"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"OV3D-Bench is a diagnostic benchmark for open-vocabulary monocular 3D detection, evaluating localization, semantic robustness, and cross-domain transfer across seven datasets.","whyItMatters":"It reveals that open-vocabulary semantics is the primary bottleneck, providing a decoupled evaluation axis for detector diagnosis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e51109fc713620a3a78a1d1190632650d677b085e585b55a4111434d1bb235a0"},"motivation":"Open-vocabulary monocular 3D detectors report strong in-domain performance, but each evaluates under a different protocol, several rely on per-image category oracles unavailable at deployment, and all collapse geometry and semantics into a single AP metric.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"OpenSUN3D workshop at ECCV'26","evidence":"Accepted to OpenSUN3D workshop at ECCV'26; benchmark is released on https://github.com/mgladkova/ov3d-bench","evidenceUrl":"https://arxiv.org/abs/2608.17110","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"OpenSUN3D workshop at ECCV'26","reviewStatus":"accepted","decisionRaw":"Accepted to OpenSUN3D workshop at ECCV'26; benchmark is released on https://github.com/mgladkova/ov3d-bench","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.17110","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to OpenSUN3D workshop at ECCV'26; benchmark is released on https://github.com/mgladkova/ov3d-bench","level":"author-claim"}]}],"publishers":[{"name":"University of Ljubljana","organizationType":"academic-lab","sourceUrl":"https://github.com/mgladkova/ov3d-bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_5b60c336f7011c62","familyId":"catalog_family_5b60c336f7011c62","name":"OVBench","oneLine":"OVBench is an online video understanding benchmark that evaluates a model's ability to perceive, memorize, and reason about real-time video streams as they unfold.","description":"OVBench is an online video understanding benchmark that evaluates a model's ability to perceive, memorize, and reason about real-time video streams as they unfold.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ovbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5b60c336f7011c62"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ovbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ovbench","url":"https://llm-stats.com/benchmarks/ovbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","video","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_ovearth-bench_169abada","familyId":"bmf_d6371bc0182b","name":"OVEarth-Bench","oneLine":"Evaluates open-vocabulary Earth observation models on category breadth and query diversity, covering mask and box localization across vocabulary, referring, and reasoning queries under a unified zero-shot protocol.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.27278","pdf":"https://arxiv.org/pdf/2607.27278","project":"https://earth-insights.github.io/OVEarth-bench","code":"https://github.com/earth-insights/OVEarth-bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.27278"},"evidence":{"snippet":"To fill this gap, we introduce OVEarth-Bench, which extends existing evaluation in two directions: category breadth, through broad hierarchical category coverage with positive and negative expressions, and query diversity, through vocabulary, referring, and reasoning queries.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":15,"hfDailySubmittedAt":"2026-07-30T00:00:00.000Z","githubStars":12,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27278"},"ranking":{"90d":{"score":42,"rank":137,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates open-vocabulary Earth observation models on category breadth and query diversity, covering mask and box localization across vocabulary, referring, and reasoning queries under a unified zero-shot protocol.","whyItMatters":"Existing EO benchmarks cover limited categories and query forms, making it hard to gauge real-world capability. This benchmark provides a broader, more diverse evaluation to compare general and EO-specific models on open-vocabulary localization.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"853b814d4a592b1b15804629d2764233a3159622e1246be4e448e1bd3f2d1769"},"motivation":"Open-vocabulary Earth observation (EO) aims to localize geospatial concepts specified in natural language rather than a fixed label set.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27278","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Earth Insights","organizationType":"academic-lab","sourceUrl":"https://earth-insights.github.io/OVEarth-bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_ovibench_2e515c0e","familyId":"bmf_007a9bd03766","name":"OVIBench","oneLine":"Evaluates VLMs on online video question answering under user interruptions, with offline simulation and multi-dimensional metrics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-26","recognitionConfidence":0.65,"links":{"report":"https://arxiv.org/abs/2608.22279","pdf":"https://arxiv.org/pdf/2608.22279","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To address this gap, we formulate the task of Online Video Question Answering under Interruption and introduce OVIBench, the first standardized benchmark for evaluating VLMs in this setting.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22279"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates VLMs on online video question answering under user interruptions, with offline simulation and multi-dimensional metrics.","whyItMatters":"Addresses the gap between offline benchmarks and real-time interactive video QA, which is essential for practical VLM deployment.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"d6e638351a4681cac0a45690d104520594aac927b23370b050707cb1e9fa1150"},"motivation":"Recent vision language models (VLMs) have achieved strong progress in video understanding.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark name is explicit, the paper describes standardized evaluation protocols, and a public release is announced.","canonicalNameSource":"paper_title","canonicalNameEvidence":"OVIBench: Benchmarking Online Video Question Answering under Interruption"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.22279","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":45,"confidence":"Medium","horizon":"7d","reason":"Interruption handling in video QA is a novel and practically relevant topic likely to attract moderate interest from VLM researchers."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_ovo-s-bench_86757d1b","familyId":"bmf_0f3bc6059546","name":"OVO-S-Bench","oneLine":"OVO-S-Bench evaluates streaming spatial intelligence in MLLMs with 1,680 human-annotated questions across four abstraction levels, using streaming prefixes and evidence intervals.","area":"Multimodal","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Automotive"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03890","pdf":"https://arxiv.org/pdf/2606.03890","project":"https://internlm.github.io/OVO-S-Bench/","code":"https://github.com/InternLM/OVO-S-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.03890"},"evidence":{"snippet":"We introduce OVO-S-Bench, a fully human-annotated benchmark for streaming spatial intelligence, comprising 1,680 questions over 348 source videos.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":31,"hfDailySubmittedAt":null,"githubStars":54,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03890"},"ranking":{"90d":{"score":54,"rank":40,"coverage":0.7,"confidence":"Medium"}},"description":"OVO-S-Bench evaluates streaming spatial intelligence in MLLMs with 1,680 human-annotated questions across four abstraction levels, using streaming prefixes and evidence intervals.","whyItMatters":"Addresses the gap in evaluating spatial reasoning from continuous egocentric streams, providing a standardized benchmark with human annotation and clear evaluation protocols.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"793c04437aa772c94e4348fc66818bbd073cbb74c4760c56d616a826c0105775"},"motivation":"Multimodal agents in robotics, AR, and autonomous driving must reason about places and layouts from continuous egocentric streams, often using evidence outside the current view.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 Main Conference","evidence":"Accepted to EMNLP 2026 Main Conference. 55 pages, 12 figures, 20 tables. Project page: https://internlm.github.io/OVO-S-Bench/","evidenceUrl":"https://arxiv.org/abs/2606.03890","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"EMNLP 2026 Main Conference","reviewStatus":"accepted","decisionRaw":"Accepted to EMNLP 2026 Main Conference. 55 pages, 12 figures, 20 tables. Project page: https://internlm.github.io/OVO-S-Bench/","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.03890","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to EMNLP 2026 Main Conference. 55 pages, 12 figures, 20 tables. Project page: https://internlm.github.io/OVO-S-Bench/","level":"author-claim"}]}],"publishers":[{"name":"InternLM","organizationType":"academic-lab","sourceUrl":"https://github.com/InternLM/OVO-S-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_9c94cb5745dfe63b","familyId":"catalog_family_9c94cb5745dfe63b","name":"OVOBench","oneLine":"OVOBench (Online Video Online Benchmark) evaluates streaming video understanding, testing a model's ability to perceive and respond to video content in real time as it unfolds.","description":"OVOBench (Online Video Online Benchmark) evaluates streaming video understanding, testing a model's ability to perceive and respond to video content in real time as it unfolds.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ovobench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9c94cb5745dfe63b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ovobench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ovobench","url":"https://llm-stats.com/benchmarks/ovobench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","video","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_p3b3_5a9c84e4","familyId":"bmf_35600c5fa04e","name":"P3B3","oneLine":"P3B3 is an expert-curated benchmark of conversational prompts for evaluating variety bias and controllability in LLMs across European and Brazilian Portuguese.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.16753","pdf":"https://arxiv.org/pdf/2606.16753","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.16753"},"evidence":{"snippet":"To address this gap, we introduce P3B3, an expert-curated language variety agnostic benchmark of conversational prompts, along with an evaluation framework for measuring variety bias and controllability.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16753"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"P3B3 is an expert-curated benchmark of conversational prompts for evaluating variety bias and controllability in LLMs across European and Brazilian Portuguese.","whyItMatters":"Evaluates regional variety bias in Portuguese LLMs, addressing underrepresentation and controllability gaps.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"59b0a456a8458ea20c319288e400dc669f34c77ec29b3087ec09ef53886424ea"},"motivation":"As Large Language Models (LLMs) become embedded in everyday communication, capturing regional linguistic variation is essential for reliable and equitable language use.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"MeLLM Workshop at ACL 2026","evidence":"Accepted at MeLLM Workshop at ACL 2026","evidenceUrl":"https://arxiv.org/abs/2606.16753","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"MeLLM Workshop at ACL 2026","reviewStatus":"accepted","decisionRaw":"Accepted at MeLLM Workshop at ACL 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.16753","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at MeLLM Workshop at ACL 2026","level":"author-claim"}]}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_p3d-bench_31251204","familyId":"bmf_15e7f389af7b","name":"P3D-Bench","oneLine":"P3D-Bench evaluates multimodal large language models on parametric 3D generation from text, image, and assembly specifications, scoring executability, geometric fidelity, topology, text-grounded constraints, multiview semantic alignment, and part-level structure.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Geometric reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11152","pdf":"https://arxiv.org/pdf/2606.11152","project":"https://spatiaos.github.io/projects/P3D-Bench","code":"https://github.com/SpatiaOS/P3D-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.11152"},"evidence":{"snippet":"We introduce P3D-Bench, a benchmark for parametric 3D generation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":"2026-06-15T00:00:00.000Z","githubStars":50,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.11152"},"ranking":{"90d":{"score":49,"rank":69,"coverage":0.7,"confidence":"Medium"}},"description":"P3D-Bench evaluates multimodal large language models on parametric 3D generation from text, image, and assembly specifications, scoring executability, geometric fidelity, topology, text-grounded constraints, multiview semantic alignment, and part-level structure.","whyItMatters":"Existing benchmarks rarely evaluate 3D modeling through code, which requires geometric precision and assembly consistency, not just runnable code. P3D-Bench provides a unified protocol to assess structural understanding and precise geometry, which is critical for models generating parametric 3D programs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a04a4e956a5aa5a993864106b3ff28929e0d91be761661e01f2ae597aac799d2"},"motivation":"Multimodal large language models can write code to produce complex programs as well as use programs to do 3D modeling, which opens up a new avenue for 3D generation powered by their priors, world knowledge and reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11152","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SpatiaOS","organizationType":"academic-lab","sourceUrl":"https://github.com/SpatiaOS/P3D-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_pace-bench_0530dc3f","familyId":"bmf_91838779e882","name":"PACE-Bench","oneLine":"PACE-Bench is a simulator-grounded benchmark with 144 source-to-target adaptation pairs across six physics domains. Each pair presents a code-driven design that succeeds in a source environment but fails in a mutated target environment, and agents must iteratively adapt the design using diagnostic sandbox feedback within a limited attempt budget.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Self-Evolution","Agents"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.14441","pdf":"https://arxiv.org/pdf/2608.14441","project":null,"code":"https://github.com/thunlp/PACE-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.14441"},"evidence":{"snippet":"To address this gap, we introduce PACE-Bench (Physics Adaptation via Code Evolution), a simulator-grounded benchmark of 144 source-to-target adaptation pairs across six physics domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":28,"hfDailySubmittedAt":"2026-08-18T00:00:00.000Z","githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14441"},"ranking":{"30d":{"score":41,"rank":47,"coverage":0.85,"confidence":"High"},"90d":{"score":39,"rank":162,"coverage":0.7,"confidence":"Medium"}},"description":"PACE-Bench is a simulator-grounded benchmark with 144 source-to-target adaptation pairs across six physics domains. Each pair presents a code-driven design that succeeds in a source environment but fails in a mutated target environment, and agents must iteratively adapt the design using diagnostic sandbox feedback within a limited attempt budget.","whyItMatters":"Existing self-evolving agent evaluations assume fixed execution conditions and do not test recovery after environmental shifts. PACE-Bench provides a repeatable protocol to assess an agent's ability to adapt to changing physics, offering insight into the reliability of different self-evolving methods.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"190109402f0e67b5f2d47cd58eab25ad418d453350fc095b2733be22d22423bf"},"motivation":"Self-evolving agents improve future behavior from interaction experience, yet existing evaluations typically optimize under fixed execution conditions and do not test recovery after those conditions change.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14441","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"THUNLP","organizationType":"academic-lab","sourceUrl":"https://github.com/thunlp/PACE-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_pacute_04643d6a","familyId":"bmf_e2d2cfe524a8","name":"PACUTE","oneLine":"PACUTE is a diagnostic benchmark of 4,600 tasks evaluating morphological understanding in Filipino, covering six compositional levels from morpheme decomposition to syllabification.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15144","pdf":"https://arxiv.org/pdf/2606.15144","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.15144"},"evidence":{"snippet":"We introduce PACUTE, a diagnostic benchmark of 4,600 tasks designed to evaluate morphological understanding in Filipino, a language characterized by productive infixation, reduplication, and diacritic-driven lexical distinctions that are typically absent from written text.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15144"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PACUTE is a diagnostic benchmark of 4,600 tasks evaluating morphological understanding in Filipino, covering six compositional levels from morpheme decomposition to syllabification.","whyItMatters":"Standard tokenizers obscure character-level and morphological structure, particularly for languages with non-concatenative morphology. This benchmark localizes where morphological understanding breaks down, distinguishing character access from productive composition, which informs tokenizer and model design for low-resource languages.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d402990649c1042705b9b5ec40901b209d1aac7afe5aa902bc0e4af10aee0853"},"motivation":"Large language models (LLMs) process text as sequences of subword tokens, which can obscure the character-level and morphological structure that underlies word formation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15144","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_paintbench_3ad5e836","familyId":"bmf_3505fc50bfea","name":"PaintBench","oneLine":"PaintBench is a procedurally generated benchmark for precise visual editing, covering 20 tasks across geometric, structural, color, and symbolic categories. Stable scoring uses pixel-level mIoU with fixed seeds.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning","Robot manipulation"],"topics":["Robotics","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00188","pdf":"https://arxiv.org/pdf/2606.00188","project":"https://paintbench.github.io/","code":"https://github.com/PaintBench/PaintBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.00188"},"evidence":{"snippet":"To probe this challenge, we introduce PaintBench, a dynamically scalable benchmark targeting 20 fundamental precise visual editing operations across four categories: geometric transformation, structural manipulation, color change, and symbolic reasoning.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00188"},"ranking":{},"description":"PaintBench is a procedurally generated benchmark for precise visual editing, covering 20 tasks across geometric, structural, color, and symbolic categories. Stable scoring uses pixel-level mIoU with fixed seeds.","whyItMatters":"Procedural generation allows contamination-resistant evaluation of precise editing capabilities, with deterministic scoring that avoids human or LLM bias.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3b9ce3e72b51dc69a158da08902c51bd2b3e1556ac4ed7da93c39ad4297e7cd3"},"motivation":"While current multimodal models are proficient at open-ended visual editing, executing precise single-answer edits remains an important obstacle.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00188","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"NYU","organizationType":"academic-lab","sourceUrl":"https://github.com/PaintBench/PaintBench","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_pair-bench_c3cd8d09","familyId":"bmf_7121e1a7ef1c","name":"PAIR-Bench","oneLine":"PAIR-Bench evaluates code improvement by transforming incorrect programs into more correct ones through feedback-guided refinement. It uses progressive hinting with failure-region and hint-depth controls to measure repair of targeted failures, generalization beyond the hint, preservation of correct behavior, and required assistance.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.01360","pdf":"https://arxiv.org/pdf/2607.01360","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.01360"},"evidence":{"snippet":"We introduce PAIR-Bench, a progressive and adaptive benchmark for evaluating code improvement: transforming an incorrect or incomplete program into a more correct one through feedback-guided refinement.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01360"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PAIR-Bench evaluates code improvement by transforming incorrect programs into more correct ones through feedback-guided refinement. It uses progressive hinting with failure-region and hint-depth controls to measure repair of targeted failures, generalization beyond the hint, preservation of correct behavior, and required assistance.","whyItMatters":"Traditional binary pass/fail metrics miss partial progress and refinement trajectories. PAIR-Bench provides finer-grained, progressive metrics to assess how LLMs improve code through feedback, offering practical value for developing and selecting models for code improvement tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7ff056914072d0fa9354c5d0518a824348b6b0efb0313c5b220eb8259131b4de"},"motivation":"Large language models (LLMs) are typically evaluated on code generation and program repair using binary functional correctness: a generated program or patch either passes or fails a test suite.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01360","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_pal-bench_e5d27a42","familyId":"bmf_645decf62e6a","name":"PAL-Bench","oneLine":"PAL-Bench evaluates evidence-grounded profile reconstruction from longitudinal personal albums, scoring agents on owner facts, identities, and relations with a seven-metric protocol.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.16175","pdf":"https://arxiv.org/pdf/2606.16175","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.16175"},"evidence":{"snippet":"We introduce PAL-Bench, a controlled benchmark for evidence-grounded reconstruction under a public-record contract.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16175"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PAL-Bench evaluates evidence-grounded profile reconstruction from longitudinal personal albums, scoring agents on owner facts, identities, and relations with a seven-metric protocol.","whyItMatters":"Existing benchmarks test sub-problems of multimodal understanding, but PAL-Bench addresses the gap in album-scale reconstruction with social identity binding and evidence citation, offering a controlled public-record contract for evaluating perceptual entity resolution and multimodal integration.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d51d5404520fd53b33f34db135127f437a7ce01430b90c28ba184dc277b194bb"},"motivation":"Longitudinal personal albums are weak-schema multimodal databases: noisy perceptual records whose key facts require joins across faces, text, timestamps, locations, and repeated events.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.16175","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_d2e6584907a815aa","familyId":"catalog_family_d2e6584907a815aa","name":"PaperBench","oneLine":"PaperBench is a benchmark for evaluating AI agents on their ability to replicate research papers. It tests models on complex, multi-step workflows involving code implementation, experimentation, and reproducing scientific results from academic publications.","description":"PaperBench is a benchmark for evaluating AI agents on their ability to replicate research papers. It tests models on complex, multi-step workflows involving code implementation, experimentation, and reproducing scientific results from academic publications.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Reasoning","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.8","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d2e6584907a815aa"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/paperbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/paperbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"paperBench","url":"https://benchlm.ai/benchmarks/paperbench","paperUrl":"https://qwen.ai/blog?id=qwen3.8","year":"2026","fullName":"PaperBench","format":"Long-horizon agent evaluation","tasks":"AI research-paper reproduction","successorKey":null},{"catalog":"llm-stats","sourceId":"paperbench","url":"https://llm-stats.com/benchmarks/paperbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","reasoning","agents","code"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_paraguibench_bf12bca6","familyId":"bmf_f889462186a8","name":"ParaGUIBench","oneLine":"ParaGUIBench aims to benchmark parallel execution and coordination of multiple GUI agents across separate desktop instances, with a dataset of 233 tasks and efficiency metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.22689","pdf":"https://arxiv.org/pdf/2607.22689","project":null,"code":"https://github.com/pkgunboat/ParaGUIBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.22689"},"evidence":{"snippet":"To close this gap, we introduce ParaGUIBench, to our knowledge, the first benchmark dedicated to parallel execution and coordination of multiple GUI agents on separate desktop instances.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.22689"},"ranking":{"90d":{"score":37,"rank":173,"coverage":0.55,"confidence":"Low"}},"description":"ParaGUIBench aims to benchmark parallel execution and coordination of multiple GUI agents across separate desktop instances, with a dataset of 233 tasks and efficiency metrics.","whyItMatters":"Could enable evaluation of parallel GUI agent coordination, potentially improving efficiency and success on long-horizon tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"db766832b3fcbdc4f443aec96d5d0666775212d8136c05a04af97e35869b8a4c"},"motivation":"Graphical user interface (GUI) agents are systems powered by large multimodal models (LMMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.22689","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_parambench_25e30479","familyId":"bmf_00f78109a13a","name":"ParamBench","oneLine":"ParamBench is a benchmark for evaluating LLM tool call parameter generation, built from real cloud-network APIs with difficulty tiers and exact match metrics. It is used to evaluate the proposed probe-guided training framework.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03071","pdf":"https://arxiv.org/pdf/2608.03071","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03071"},"evidence":{"snippet":"To support systematic evaluation, we release ParamBench, a benchmark built from real cloud-network APIs that categorizes every instance into five difficulty levels according to parameter nesting depth, cross-parameter dependencies, and the reasoning required to derive values from earlier calls.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03071"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ParamBench is a benchmark for evaluating LLM tool call parameter generation, built from real cloud-network APIs with difficulty tiers and exact match metrics. It is used to evaluate the proposed probe-guided training framework.","whyItMatters":"Tool call parameter correctness is critical for execution yet understudied. ParamBench provides a systematic evaluation of parameter generation across difficulty levels, showing large improvements from probe-guided methods.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"86610886987ac92dc1bf815d5db268a7fa427bc436c85d171ce3f15d019d96de"},"motivation":"Large language model agents derive much of their capability from tool use.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03071","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_parapairaudiobench_e7d0c4ed","familyId":"bmf_7892b914cd26","name":"ParaPairAudioBench","oneLine":"ParaPairAudioBench evaluates LALMs as judges for paralinguistic speech across five dimensions with 5,175 audio pairs.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SD"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.24648","pdf":"https://arxiv.org/pdf/2606.24648","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.24648"},"evidence":{"snippet":"We introduce ParaPairAudioBench, a pairwise benchmark of 5,175 audio pairs across five paralinguistic dimensions: Style, Rate, Emphasis, Age, and Gender.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24648"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ParaPairAudioBench evaluates LALMs as judges for paralinguistic speech across five dimensions with 5,175 audio pairs.","whyItMatters":"Targets fine-grained paralinguistic distinctions that prior benchmarks overlook, enabling calibration-aware assessment of judge reliability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a5089523d7a6a8813f7a7aeb60994920afc28666734033e8eabb5f1f552f6607"},"motivation":"Large Audio-Language Models (LALMs) have been widely used as judge models for the automatic evaluation of generated speech.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"Interspeech 2026","evidence":"Accepted to Interspeech 2026","evidenceUrl":"https://arxiv.org/abs/2606.24648","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"Interspeech 2026","reviewStatus":"accepted","decisionRaw":"Accepted to Interspeech 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.24648","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to Interspeech 2026","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_parasgb_cd57a968","familyId":"bmf_3039ff119f9f","name":"ParasGB","oneLine":"ParasGB evaluates graph neural network predictions of parasitic capacitance and resistance on circuit graphs from analog/mixed-signal designs, with node-level ground capacitance, edge-level resistance, and edge-level coupling capacitance tasks.","area":"Language & Knowledge","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Semiconductors","Software & Cloud","Manufacturing"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.23225","pdf":"https://arxiv.org/pdf/2607.23225","project":null,"code":"https://github.com/ShenShan123/ParasGB.git","data":null,"hfPaper":"https://huggingface.co/papers/2607.23225"},"evidence":{"snippet":"To address this gap, we introduce ParasGB, the first open-source benchmark suite for pre-layout parasitic parameter prediction on circuit graphs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23225"},"ranking":{"90d":{"score":23,"rank":367,"coverage":0.55,"confidence":"Low"}},"description":"ParasGB evaluates graph neural network predictions of parasitic capacitance and resistance on circuit graphs from analog/mixed-signal designs, with node-level ground capacitance, edge-level resistance, and edge-level coupling capacitance tasks.","whyItMatters":"ParasGB addresses the lack of public high-fidelity RC benchmarks for early parasitic estimation, enabling reproducible evaluation and development of GNN-based models for parasitic-aware design.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1802dfa3f3a0074748f8f7b7fa90cd963ce4e69b83906bf0eb4f5e84a3351d3f"},"motivation":"As chip manufacturing processes advance to deep submicron nodes, parasitic interconnect effects increasingly dominate the performance of analog and mixed-signal (AMS) circuits and often lead to costly layout iterations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.23225","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ShenShan123","organizationType":"community","sourceUrl":"https://github.com/ShenShan123/ParasGB.git","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"specific"},{"id":"bm_pasbench-video_ab3b3eb4","familyId":"bmf_6df4c19f80cb","name":"PaSBench-Video","oneLine":"PaSBench-Video is a 740-video benchmark for proactive safety warning with 481 risk and 259 no-risk videos across driving, healthcare, daily life, and industrial production. Annotations include frame-level risk onset and accident boundaries. Models must process video causally and output a warning that is temporally calibrated and content-correct.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02443","pdf":"https://arxiv.org/pdf/2606.02443","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02443"},"evidence":{"snippet":"We present PaSBench-Video, a 740-video benchmark with 481 risk and 259 no-risk videos across four domains: driving, healthcare, daily life, and industrial production.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02443"},"ranking":{},"description":"PaSBench-Video is a 740-video benchmark for proactive safety warning with 481 risk and 259 no-risk videos across driving, healthcare, daily life, and industrial production. Annotations include frame-level risk onset and accident boundaries. Models must process video causally and output a warning that is temporally calibrated and content-correct.","whyItMatters":"This benchmark addresses a gap in evaluating video MLLMs for real-time safety monitoring, emphasizing temporal calibration and false-positive control on safe scenes. It provides a standardized protocol to compare models' ability to issue timely warnings, which is critical for deployment in safety-sensitive applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8c790c25e946c2839989c8470d79faece82637cd7edcdd4911882a0fce6b870a"},"motivation":"Between the first visible sign of danger and the moment an accident occurs, there is often a window where intervention remains possible.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02443","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_past-bench_b04dad58","familyId":"bmf_d6508586cac7","name":"PAST-Bench","oneLine":"PAST-Bench evaluates recursive self-improvement in personal AI agents by testing whether retained experience improves performance on future tasks. It spans 26 scenarios and 204 episodes across memory, procedural reuse, information gathering, and update capabilities, with matched persistence on/off controls.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Self-Evolution","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04003","pdf":"https://arxiv.org/pdf/2608.04003","project":null,"code":"https://github.com/Gen-Verse/PAST-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.04003"},"evidence":{"snippet":"We introduce PAST-Bench, a benchmark designed to isolate this question.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":34,"hfDailySubmittedAt":"2026-08-05T00:00:00.000Z","githubStars":24,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04003"},"ranking":{"30d":{"score":53,"rank":17,"coverage":0.85,"confidence":"High"},"90d":{"score":49,"rank":75,"coverage":0.7,"confidence":"Medium"}},"description":"PAST-Bench evaluates recursive self-improvement in personal AI agents by testing whether retained experience improves performance on future tasks. It spans 26 scenarios and 204 episodes across memory, procedural reuse, information gathering, and update capabilities, with matched persistence on/off controls.","whyItMatters":"Whether personal agents actually improve from retained experience has not been systematically tested. PAST-Bench provides a controlled benchmark to measure and attribute cross-session improvement, distinguishing capability gains from pathway evidence.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0ef8aa96463f20467ad56ab00c20214e33f953a38a56dd9a61d0b003234a832f"},"motivation":"Recursive self-improvement requires agents to turn accumulated experience into better future behavior.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04003","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Gen-Verse","organizationType":"community","sourceUrl":"https://github.com/Gen-Verse/PAST-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_patchfusebench_d8b0e871","familyId":"bmf_060068620f54","name":"PatchFuseBench","oneLine":"PatchFuseBench is a fixed-pool benchmark for evaluating repair candidate fusion, built from existing SWE-bench Verified, SWE-bench Multilingual, and Defects4J candidate patches. The benchmark pools candidate patches for 500 bugs on SWE-bench Verified, 300 on Multilingual, and 371 on Defects4J, and evaluates methods that fuse or select a final patch.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.01597","pdf":"https://arxiv.org/pdf/2607.01597","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.01597"},"evidence":{"snippet":"To evaluate this setting, we build PatchFuseBench, a fixed-pool benchmark covering SWE-bench Verified, SWE-bench Multilingual, and Defects4J candidate patches.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01597"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PatchFuseBench is a fixed-pool benchmark for evaluating repair candidate fusion, built from existing SWE-bench Verified, SWE-bench Multilingual, and Defects4J candidate patches. The benchmark pools candidate patches for 500 bugs on SWE-bench Verified, 300 on Multilingual, and 371 on Defects4J, and evaluates methods that fuse or select a final patch.","whyItMatters":"It addresses the pass@k-to-pass@1 gap in code repair, where candidate pools may contain correct patches but selection remains challenging. The benchmark provides a controlled setting to compare post-generation patch selection and fusion methods.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a28f0649ea596e6fecd5fc741ddafdd391dd31066feb15042512db4c67884169"},"motivation":"Modern LLM coding agents are commonly evaluated using pass@k, but developers typically apply a single final patch in real-world settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01597","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_pathagentbench_70491a6b","familyId":"bmf_70918a81327d","name":"PathAgentBench","oneLine":"PathAgentBench evaluates vision-language models on whole-slide pathology images across four capabilities: image-to-text matching, text-to-image retrieval, diagnostic-region localization, and multi-scale reasoning. It includes 1,822 TCGA WSIs and 17,135 diagnostic paths.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning","Information retrieval"],"topics":["Multimodal","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19261","pdf":"https://arxiv.org/pdf/2607.19261","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19261"},"evidence":{"snippet":"We introduce PathAgentBench, a benchmark for evaluating evidence-seeking vision-language models (VLMs) across four complementary capabilities: image-to-text matching for evidence interpretation, text-to-image retrieval for evidence verification, diagnostic-region localization for evidence acquisition, and multi-scale reasoning for evidence integration.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19261"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PathAgentBench evaluates vision-language models on whole-slide pathology images across four capabilities: image-to-text matching, text-to-image retrieval, diagnostic-region localization, and multi-scale reasoning. It includes 1,822 TCGA WSIs and 17,135 diagnostic paths.","whyItMatters":"Most pathology benchmarks use pre-cropped patches, not whole-slide exploration. PathAgentBench provides a unified framework with annotated paths, revealing a significant gap in evidence acquisition and supporting progress in autonomous WSI diagnosis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"164a62c1666c683614d9776c3857d004743532095c87209c9cd9c0e1c75bad41"},"motivation":"Whole-slide image (WSI) diagnosis requires identifying diagnostically relevant regions, examining them across magnifications, and integrating multi-scale evidence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19261","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Search & Retrieval"],"domainScope":"specific"},{"id":"catalog_a287489a87c78047","familyId":"catalog_family_a287489a87c78047","name":"PathMCQA","oneLine":"PathMMU is a massive multimodal expert-level benchmark for understanding and reasoning in pathology, containing 33,428 multimodal multi-choice questions and 24,067 images validated by seven pathologists. It evaluates Large Multimodal Models (LMMs) performance on pathology tasks, with the top-performing model GPT-4V achieving only 49.8% zero-shot performance compared to 71.8% for human pathologists.","description":"PathMMU is a massive multimodal expert-level benchmark for understanding and reasoning in pathology, containing 33,428 multimodal multi-choice questions and 24,067 images validated by seven pathologists. It evaluates Large Multimodal Models (LMMs) performance on pathology tasks, with the top-performing model GPT-4V achieving only 49.8% zero-shot performance compared to 71.8% for human pathologists.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Healthcare","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/pathmcqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a287489a87c78047"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/pathmcqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"pathmcqa","url":"https://llm-stats.com/benchmarks/pathmcqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","healthcare","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_pathoargus-bench_2bed53e5","familyId":"bmf_9d5dfa0f7014","name":"PathoArgus-Bench","oneLine":"Evaluates evidence-grounded visual reasoning over complete gigapixel whole-slide and multi-slide pathology cases.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Medical image reasoning","Long-context reasoning","Evidence grounding"],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.17607","pdf":"https://arxiv.org/pdf/2608.17607","project":null,"code":null,"data":"https://huggingface.co/datasets/liubw/PathoArgus-Bench","hfPaper":"https://huggingface.co/papers/2608.17607"},"evidence":{"snippet":"We introduce PathoArgus-Bench, a benchmark and evaluation protocol that explicitly tests the full evidence chain: availability, accessibility, use, and responsiveness.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":74,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2608.17607"},"ranking":{"30d":{"score":49,"rank":31,"coverage":0.15,"confidence":"Low","datasetDownloadRank":22,"datasetRankPopulation":30},"90d":{"score":46,"rank":101,"coverage":0.3,"confidence":"Low","datasetDownloadRank":50,"datasetRankPopulation":66}},"description":"PathoArgus-Bench evaluates evidence-grounded visual reasoning in whole-slide pathology. It comprises 22,078 multiple-choice questions from 4,913 patients across 15 TCGA projects, testing availability, accessibility, use, and responsiveness of evidence under a fixed reader budget.","whyItMatters":"Addresses the gap where final answer accuracy is insufficient to establish evidence grounding in pathology AI. Provides a protocol to assess whether models truly use supplied tissue evidence, offering a more rigorous evaluation for clinical deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"459e70fb59ee84661c78fd4108f27802d708370fbf5061622b60b457c6dfdb2f"},"motivation":"Whole-slide pathology reasoning requires models to integrate gigapixel-scale visual evidence across complete case-linked slides, yet current question-answering benchmarks primarily measure final answer accuracy--a metric vulnerable to linguistic priors and benchmark regularities, and insufficient to establish that predictions are grounded in the supplied tissue.","constructionDetail":"PathoArgus-Bench measures case-level pathology reasoning over whole-slide images, including diagnostic findings, staging and exact evidence grounding.","detail":{"taskBreakdown":["Anatomic site","Histologic type","Tumor grade","Mitotic activity","Local invasion","Surgical margins","Nodal status","Pathologic staging","Cross-slide integration","Evidence grounding"],"protocol":{"tasks":"22,078 four-choice questions from 4,913 patients and 5,400 whole-slide images","primaryMetric":"Overall accuracy and QExact","version":"arXiv v1"}},"curation":{"state":"source-reviewed","reviewedAt":"2026-08-20","sources":["https://arxiv.org/abs/2608.17607","https://arxiv.org/html/2608.17607","https://huggingface.co/datasets/liubw/PathoArgus-Bench"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17607","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"score_submission","availability":{"paperStatus":"available","githubStatus":"not_found","hfDatasetStatus":"available","evaluatorStatus":"not_found","submissionStatus":"not_found"},"publishers":[{"name":"PathoArgus-Bench Team","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/datasets/liubw/PathoArgus-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception","Long Context & Memory"],"domainScope":"specific"},{"id":"bm_pathview-bench_295675e0","familyId":"bmf_8ae3ed6f0412","name":"PathView-Bench","oneLine":"PathVU is a benchmark for fine-grained multiscale visual understanding in pathology, built from 23 public datasets with human-supervised labels and spatial annotations, covering region and slide fields of view, with 14 VQA-style tasks and 308k samples.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.28318","pdf":"https://arxiv.org/pdf/2607.28318","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.28318"},"evidence":{"snippet":"We introduce PathVU, a vision-anchored benchmark for fine-grained and multiscale visual understanding in computational pathology.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.28318"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PathVU is a benchmark for fine-grained multiscale visual understanding in pathology, built from 23 public datasets with human-supervised labels and spatial annotations, covering region and slide fields of view, with 14 VQA-style tasks and 308k samples.","whyItMatters":"PathVU aims to evaluate fine-grained visual understanding in pathology MLLMs, potentially offering more detailed assessment than existing benchmarks that focus on final diagnoses.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3e923a1c3ccc600627a0c4f8447f237a4766347eddb9d236a45e9e93bbab165e"},"motivation":"Multimodal large language models (MLLMs) are increasingly used to analyze pathology images.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.28318","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_beyond-correctness-benchmarking-and-aligni_75c36d61","familyId":"bmf_fa1620ab6433","name":"PatternEval","oneLine":"PatternEval is a diagnostic benchmark with 2,415 multimodal prompts testing four response-pattern failures: chain-of-thought leakage, repetition, contradiction, and performative reasoning.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Factuality"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.12781","pdf":"https://arxiv.org/pdf/2608.12781","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce \\textbf{PatternEval}, a failure-enriched diagnostic benchmark comprising 2,415 multimodal prompts spanning visual perception and grounding, structured image understanding, and multimodal knowledge reasoning.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":35,"hfDailySubmittedAt":"2026-08-24T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12781"},"ranking":{"30d":{"score":54,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":52,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"PatternEval is a diagnostic benchmark with 2,415 multimodal prompts testing four response-pattern failures: chain-of-thought leakage, repetition, contradiction, and performative reasoning.","whyItMatters":"Evaluates response-pattern alignment between thinking and non-thinking modes in hybrid-thinking MLLMs, addressing user-facing quality beyond accuracy.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"0ffdaaa20734fa2298806ce7e50d0eda18ea3b3d2f71264baec1784996f335d9"},"motivation":"Hybrid-thinking multimodal large language models (MLLMs) allow a single model to alternate between deliberative thinking and latency-efficient non-thinking inference.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is explicitly named and introduced, and the paper presents a clear scoring protocol; however, the abstract does not mention public code or data release, though the benchmark is intended for reuse.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce \\textbf{PatternEval}, a failure-enriched diagnostic benchmark comprising 2,415 multimodal prompts spanning visual perception and grounding, structured image understanding, and multimodal knowledge reasoning."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12781","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":50,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses an emerging area of hybrid-thinking models and includes systematic failure analysis, likely to interest MLLM evaluation researchers."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_pause_e497d85d","familyId":"bmf_6210c0bf0539","name":"PAUSE","oneLine":"A user-centric benchmark for evaluating personal AI assistants in stateful, service-integrated environments, with multi-regime evaluation and user simulation for long-horizon tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27354","pdf":"https://arxiv.org/pdf/2607.27354","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.27354"},"evidence":{"snippet":"We introduce PAUSE, a user-centric benchmark for evaluating personal AI assistants in stateful, service-integrated environments.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27354"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A user-centric benchmark for evaluating personal AI assistants in stateful, service-integrated environments, with multi-regime evaluation and user simulation for long-horizon tasks.","whyItMatters":"Personal AI assistants need to handle stateful, user-configuration-aware interactions across services; this benchmark provides a framework for evaluating user-centric performance in realistic settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2e4413afc06534891930cb8cda1d3f06664b6536d382ee0fee37e3357bff47df"},"motivation":"Personal AI assistants are increasingly deployed as task-oriented, tool-augmented agents that operate within unified service environments to support everyday user activities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27354","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_pawbench_0877eb45","familyId":"bmf_d1e745e2c7d4","name":"PAWBench","oneLine":"Evaluates video generators as stochastic samplers of world dynamics across 50 scenarios, comparing the distribution of possible behaviors under identical initial observations and actions.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2608.27345","pdf":"https://arxiv.org/pdf/2608.27345","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To answer it, we formalize probabilistic alignment as a distributional criterion for world models and introduce PAWBench, a benchmark for evaluating video generators as stochastic samplers of world dynamics.","reasonCodes":["coined title prefix ending in Bench or Benchmark"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":null,"hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27345"},"ranking":{},"description":"Evaluates video generators as stochastic samplers of world dynamics across 50 scenarios, comparing the distribution of possible behaviors under identical initial observations and actions.","whyItMatters":"Current evaluations assess single-video plausibility, missing whether models recover the correct distribution of physical behavior, a requirement for reliable world modeling.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T07:23:36.749451Z","inputHash":"8690b52f9124fb32dd2a5fece1abe1edd849d60c3ec784c93b19b5f78ba99479"},"motivation":"Recent video generation models are increasingly framed as world models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T07:23:36.749451Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with explicit evaluation protocol and scoring contract; no public artifact link provided yet, but code and data are indicated for release upon acceptance.","canonicalNameSource":"abstract","canonicalNameEvidence":"introduce PAWBench, a benchmark for evaluating video generators as stochastic samplers of world dynamics"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.27345","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-30T06:59:55.242506Z"},"attentionForecast":{"score":70,"confidence":"Medium","horizon":"7d","reason":"Timely focus on distributional evaluation of world models and release announcement could draw strong interest from the video generation community."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_pcbnet_ffc56858","familyId":"bmf_759bea05f02a","name":"PCBnet","oneLine":"Printed circuit boards (PCBs) are fundamental to modern electronic systems, yet AI-driven PCB design automation remains constrained by the lack of large-scale paired schematic-netlist datasets.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":0.85,"links":{"report":"http://arxiv.org/abs/2608.27923v1","pdf":"https://arxiv.org/pdf/2608.27923v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"PCBnet provides a benchmark and data foundation for future AI-driven PCB design automation.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27923"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"Printed circuit boards (PCBs) are fundamental to modern electronic systems, yet AI-driven PCB design automation remains constrained by the lack of large-scale paired schematic-netlist datasets.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T07:23:28.296315Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"acceptance_claimed","venue":"2026 IEEE International Conference on LLM-Aided Design (ICLAD 2026)","evidence":"Accepted at the 2026 IEEE International Conference on LLM-Aided Design (ICLAD 2026)","evidenceUrl":"http://arxiv.org/abs/2608.27923v1","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"2026 IEEE International Conference on LLM-Aided Design (ICLAD 2026)","reviewStatus":"accepted","decisionRaw":"Accepted at the 2026 IEEE International Conference on LLM-Aided Design (ICLAD 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"http://arxiv.org/abs/2608.27923v1","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at the 2026 IEEE International Conference on LLM-Aided Design (ICLAD 2026)","level":"author-claim"}]}],"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_peakbench_8142057d","familyId":"bmf_3fec537d2f24","name":"PeakBench","oneLine":"PeakBench is a benchmark of executable multi-tool workflows for evaluating resource-aware tool invocation in LLM agents. It includes execution-grounded dependency annotations and measured resource profiles, with a two-part evaluation framework distinguishing logical planning from physical scheduling. Code is available on GitHub.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-25","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.24509","pdf":"https://arxiv.org/pdf/2608.24509","project":null,"code":"https://github.com/Czzzk/Staggering-the-Peaks","data":null,"hfPaper":null},"evidence":{"snippet":"To address this gap, we introduce PeakBench, a benchmark of executable multi-tool workflows with execution-grounded dependency annotations and measured resource profiles.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24509"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PeakBench is a benchmark of executable multi-tool workflows for evaluating resource-aware tool invocation in LLM agents. It includes execution-grounded dependency annotations and measured resource profiles, with a two-part evaluation framework distinguishing logical planning from physical scheduling. Code is available on GitHub.","whyItMatters":"Existing agent benchmarks overlook parallelization and resource-constrained scheduling, creating practical failure modes. PeakBench provides a testbed to diagnose resource-aware agent behavior, helping improve safe and efficient tool execution.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"60282cff8ca5502cfcd6f3e7177a1626619bb6e90e24a4b740c677419d3e3c09"},"motivation":"LLM agents increasingly solve tasks by invoking multiple tools, where parallel execution is essential for low latency but difficult to manage safely.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:08:47.245811Z","model":"deepseek-v4-pro","decisionReason":"Code link is provided; the benchmark includes a defined two-part evaluation framework with dedicated metrics.","canonicalNameSource":"paper_title","canonicalNameEvidence":"PeakBench: Benchmarking Resource-Aware Tool Invocation in LLM Agents"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24509","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":55,"confidence":"Low","horizon":"7d","reason":"Specialized topic in LLM agents; code availability may generate moderate interest among agent researchers."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_peft-arena_b7338aaa","familyId":"bmf_f5e847eb890b","name":"PEFT-Arena","oneLine":"PEFT-Arena benchmark jointly evaluates target-domain performance and retention of pretrained capabilities for parameter-efficient finetuning methods. It covers mathematical and medical reasoning as target domains and measures general capability retention on benchmarks including BBH, IFEval, and NQ with SFT and RLVR training settings.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.28819","pdf":"https://arxiv.org/pdf/2605.28819","project":"https://spherelab.ai/PEFT-Arena/","code":"https://github.com/Sphere-AI-Lab/PEFT-Arena","data":null,"hfPaper":"https://huggingface.co/papers/2605.28819"},"evidence":{"snippet":"We introduce PEFT-Arena, a benchmark that jointly measures downstream performance and general capability retention.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":9,"hfDailySubmittedAt":null,"githubStars":28,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28819"},"ranking":{},"description":"PEFT-Arena benchmark jointly evaluates target-domain performance and retention of pretrained capabilities for parameter-efficient finetuning methods. It covers mathematical and medical reasoning as target domains and measures general capability retention on benchmarks including BBH, IFEval, and NQ with SFT and RLVR training settings.","whyItMatters":"Existing PEFT evaluations focus mainly on downstream accuracy, overlooking retention of pretrained abilities. This benchmark provides a stability-plasticity perspective, enabling selection of fine-tuning methods that balance task adaptation and forgetting resistance, offering a more complete assessment for practical deployment decisions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"deabbebd645b7ebe77205d30ddd4a8cce565c608114c85c53cd71e0d4157c0bd"},"motivation":"Parameter-efficient finetuning (PEFT) has become the standard approach for adapting large language models, yet evaluations largely emphasize downstream accuracy while overlooking the retention of pretrained capabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28819","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Sphere AI Lab","organizationType":"academic-lab","sourceUrl":"https://spherelab.ai/PEFT-Arena/","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_2d3780b918bb8299","familyId":"catalog_family_2d3780b918bb8299","name":"Pencil Puzzle Bench","oneLine":"A multi-step verifiable reasoning benchmark that evaluates whether models can solve pencil puzzles with unique solutions.","description":"A multi-step verifiable reasoning benchmark that evaluates whether models can solve pencil puzzles with unique solutions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2603.02119","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2d3780b918bb8299"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/ppbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"ppBench","url":"https://benchlm.ai/benchmarks/ppbench","paperUrl":"https://arxiv.org/abs/2603.02119","year":"2026","fullName":"Pencil Puzzle Bench","format":"Direct and agentic puzzle solve rate","tasks":"300 evaluation puzzles","successorKey":null}],"catalogCategories":["reasoning"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_do-large-language-models-perform-well-on-c_54280d5e","familyId":"bmf_6d385d6932a3","name":"Peony","oneLine":"Evaluates poetic logic in modern Chinese poetry through four tasks across stanza, line, and imagery levels using six mainstream LLMs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-22","firstSeenAt":"2026-08-25","recognitionConfidence":0.6,"links":{"report":"http://arxiv.org/abs/2608.21827v1","pdf":"https://arxiv.org/pdf/2608.21827v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To address this gap, we propose Peony, the first benchmark specifically designed for evaluating the poetic logic of modern Chinese poetry.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.21827"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates poetic logic in modern Chinese poetry through four tasks across stanza, line, and imagery levels using six mainstream LLMs.","whyItMatters":"Addresses the lack of benchmarks for literary reasoning, specifically the holistic understanding required for modern Chinese poetry beyond superficial semantics.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:28:56.873215Z","inputHash":"30d9768c401ef43dea65623210e649b9dd31ce8cced5608939e6aa1c785bbbea"},"motivation":"Large Language Models (LLMs) have achieved significant progress across a wide range of natural language processing (NLP) tasks, yet their ability to understand literary texts, particularly modern Chinese poetry, remains largely unexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T23:28:56.873215Z","model":"deepseek-v4-pro","decisionReason":"Insufficient evidence of a public release path or formal benchmark infrastructure; only promise of data/code availability without concrete artifacts.","canonicalNameSource":"abstract","canonicalNameEvidence":"we propose Peony, the first benchmark specifically designed for evaluating the poetic logic of modern Chinese poetry"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.21827v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":40,"confidence":"Low","horizon":"7d","reason":"Novelty in evaluating poetic logic in Chinese poetry may attract interest, but lack of public artifacts and niche scope limit early attention."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_perceptionbench_27ae03ae","familyId":"bmf_885ad8ebb2ad","name":"PerceptionBench","oneLine":"PerceptionBench evaluates atomic visual perception in MLLMs with 3,000 verified questions isolating ten perceptual capabilities, based on an error taxonomy from 42 benchmarks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.24957","pdf":"https://arxiv.org/pdf/2607.24957","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.24957"},"evidence":{"snippet":"We introduce PerceptionBench, a benchmark specifically designed to evaluate the atomic visual perception capabilities of Multimodal Large Language Models (MLLMs).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":19,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24957"},"ranking":{"90d":{"score":50,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"PerceptionBench evaluates atomic visual perception in MLLMs with 3,000 verified questions isolating ten perceptual capabilities, based on an error taxonomy from 42 benchmarks.","whyItMatters":"Addresses the need for a capability-level standard to diagnose visual perception boundaries, showing that current MLLMs remain below 60% accuracy.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1dbc8e54ea45881534d104682db2e2e19d0ded65ba923bbca03fd4741fd614d7"},"motivation":"We introduce PerceptionBench, a benchmark specifically designed to evaluate the atomic visual perception capabilities of Multimodal Large Language Models (MLLMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24957","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"perceptionBench","url":"https://benchlm.ai/benchmarks/perceptionbench","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"PerceptionBench (Internal)","format":"Internal evaluation score","tasks":"Internal atomic visual-perception tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"perceptionbench","url":"https://llm-stats.com/benchmarks/perceptionbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","vision"],"catalogModelCount":2,"catalogStarCount":0},{"id":"catalog_9453ef18dc314e47","familyId":"catalog_family_9453ef18dc314e47","name":"PerceptionTest","oneLine":"A novel multimodal video benchmark designed to evaluate perception and reasoning skills of pre-trained models across video, audio, and text modalities. Contains 11.6k real-world videos (average 23 seconds) filmed by participants worldwide, densely annotated with six types of labels. Focuses on skills (Memory, Abstraction, Physics, Semantics) and reasoning types (descriptive, explanatory, predictive, counterfactual). Shows significant performance gap between human baseline (91.4%) and state-of-the-art video QA models (46.2%).","description":"A novel multimodal video benchmark designed to evaluate perception and reasoning skills of pre-trained models across video, audio, and text modalities. Contains 11.6k real-world videos (average 23 seconds) filmed by participants worldwide, densely annotated with six types of labels. Focuses on skills (Memory, Abstraction, Physics, Semantics) and reasoning types (descriptive, explanatory, predictive, counterfactual). Shows significant performance gap between human baseline (91.4%) and state-of-the-art video QA models (46.2%).","area":"Mathematical Reasoning","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Multimodal","Physics","Reasoning","Spatial Reasoning","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/perceptiontest","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9453ef18dc314e47"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/perceptiontest"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"perceptiontest","url":"https://llm-stats.com/benchmarks/perceptiontest","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","physics","reasoning","spatial reasoning","video","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"bm_perfopt-bench_3a913e78","familyId":"bmf_1425682db99d","name":"PERFOPT-Bench","oneLine":"PERFOPT-Bench evaluates coding agents on software performance optimization tasks, requiring profiling, diagnosing bottlenecks, editing code while preserving correctness, and verifying reproducible speedups. Scoring includes hidden correctness tests, verified-speedup measurement, and trajectory-level audit.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.07744","pdf":"https://arxiv.org/pdf/2607.07744","project":"https://anonymous.4open.science/r/Dataset-D3CC","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.07744"},"evidence":{"snippet":"We introduce PERFOPT-Bench, a benchmark for evaluating this full performance-engineering loop.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.07744"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"PERFOPT-Bench evaluates coding agents on software performance optimization tasks, requiring profiling, diagnosing bottlenecks, editing code while preserving correctness, and verifying reproducible speedups. Scoring includes hidden correctness tests, verified-speedup measurement, and trajectory-level audit.","whyItMatters":"Fills the gap in benchmarks focusing on performance engineering rather than functional correctness, measuring practical speedups on real execution targets and addressing issues like shortcut exploitation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d23f95d9142ead8dcf00fe5ed0553c972cf1a01323de436ebea31c7b3f7d2da7"},"motivation":"Coding-agent benchmarks have largely measured whether agents can produce functionally correct patches, but production software also demands measurable speedups on real execution targets.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.07744","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_permembench_15a1b01c","familyId":"bmf_fb7852f323c5","name":"PerMemBench","oneLine":"PerMemBench evaluates personalized memory systems for LLM agents using multi-year, multi-domain interaction histories across 20 user personas, measuring memory retention accuracy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.25535","pdf":"https://arxiv.org/pdf/2605.25535","project":null,"code":"https://github.com/yeonjun-in/PerMemBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.25535"},"evidence":{"snippet":"We introduce PerMemBench, the first benchmark for evaluating personalized memory systems, featuring multi year, multi domain interaction histories across diverse user personas.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":45,"hfDailySubmittedAt":null,"githubStars":10,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25535"},"ranking":{},"description":"PerMemBench evaluates personalized memory systems for LLM agents using multi-year, multi-domain interaction histories across 20 user personas, measuring memory retention accuracy.","whyItMatters":"Universal memory policies waste budget on transient interactions and fail to preserve critical context. PerMemBench enables evaluation of personalization, revealing that accurate gating remains an open challenge.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f7bcccbe7d923b8a2bbed3c622d775edc3b6607c3633cd1a06c98d82b39db8b0"},"motivation":"Existing large language model (LLM) based memory systems apply universal, static policies that overlook a fundamental reality: the contexts that are worth storing in memory are different across users.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25535","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"PerMemBench team","organizationType":"academic-lab","sourceUrl":"https://github.com/yeonjun-in/PerMemBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_persliteval_aca10df9","familyId":"bmf_93ba18bfa5d3","name":"PersLitEval","oneLine":"PersLitEval is a benchmark of 4,514 Persian literature multiple-choice questions across eight categories including spelling, literary devices, grammar, vocabulary, word formation, and conceptual understanding, sourced from Konkur university entrance exam materials.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27015","pdf":"https://arxiv.org/pdf/2605.27015","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27015"},"evidence":{"snippet":"We introduce PersLitEval, a benchmark of 4,514 Persian literature multiple-choice questions across eight fine-grained categories spanning spelling, literary devices, grammar, vocabulary, word formation, and conceptual understanding, sourced from materials for the Konkur university entrance examination.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27015"},"ranking":{},"description":"PersLitEval is a benchmark of 4,514 Persian literature multiple-choice questions across eight categories including spelling, literary devices, grammar, vocabulary, word formation, and conceptual understanding, sourced from Konkur university entrance exam materials.","whyItMatters":"LLMs remain poorly evaluated on literary knowledge in non-English languages. PersLitEval provides a fine-grained evaluation of Persian literary understanding, revealing disparities across task difficulty and prompting strategies, which can guide improvements for multilingual literary competence.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e7fb8c408f5b29e53d9708c7e4f48b22f2226d11bcd0582c6fe5c17bef2dbf33"},"motivation":"Despite impressive multilingual capabilities, large language models (LLMs) remain poorly evaluated on literary knowledge in non-English languages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27015","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_personalbench_05e1f48e","familyId":"bmf_83ab15035e3e","name":"PersonalBench","oneLine":"Evaluates personalized text generation through LUAR, LLM-as-judge, and stylometrics across 50 authors and 1,000 generations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-21","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.19746","pdf":"https://arxiv.org/pdf/2608.19746","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce PersonalBench, a benchmark that evaluates inference-time personalization methods through three independent lenses: LUAR (a trained authorship verification model), an LLM-as-judge, and automated stylometrics.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.19746"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates personalized text generation through LUAR, LLM-as-judge, and stylometrics across 50 authors and 1,000 generations.","whyItMatters":"Provides a calibrated measure of authorship gap, revealing that personalization methods fail to bridge human-LLM boundary.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"0b0af6ce7cd4d4ed878dce38cdf1df9ab1bd390c36e4945d474bdc1d0015c8ec"},"motivation":"Personalized text generation aims to make LLMs write in a specific individual's style, yet existing benchmarks measure task accuracy or preference alignment rather than whether the model's output actually resembles the target author's writing.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"While the paper introduces a benchmark, it primarily supports the findings of a single paper and lacks a standalone public comparison path."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.19746","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-21T04:30:09.569171Z"},"attentionForecast":{"score":30,"confidence":"Low","horizon":"7d","reason":"Personalization is a current topic but this benchmark's lack of public data and standalone comparison may limit engagement."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_personashot_741ba748","familyId":"bmf_2db553a9d65f","name":"PersonaShot","oneLine":"PersonaShot benchmarks person-centric narrative continuity in multi-shot video generation with ~1,000 segments and 16 metrics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.16717","pdf":"https://arxiv.org/pdf/2608.16717","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.16717"},"evidence":{"snippet":"To address these limitations, we introduce PersonaShot, the first person-centric benchmark for narrative continuity in multi-shot video generation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.16717"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PersonaShot benchmarks person-centric narrative continuity in multi-shot video generation with ~1,000 segments and 16 metrics.","whyItMatters":"It addresses the gap in evaluating character coherence across video cuts, including physical and emotional state continuity.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a05db40195b8a4d66c0e57952bf73a94db951d956fb00cd93b728764310a7844"},"motivation":"Video generation is rapidly evolving from single-shot clips to multi-shot narratives, where the human character serves as the core narrative anchor.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.16717","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_personatrail_ac04b54e","familyId":"bmf_70d82a445a60","name":"PersonaTrail","oneLine":"PersonaTrail evaluates personalized web agents using realistic browsing trajectories as user history, assessing preference inference and information recall. Operates in a managed open web environment with two tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20482","pdf":"https://arxiv.org/pdf/2607.20482","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20482"},"evidence":{"snippet":"To bridge this gap, we introduce PersonaTrail, a benchmark for personalized web agents operating in a managed open web environment.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20482"},"ranking":{},"description":"PersonaTrail evaluates personalized web agents using realistic browsing trajectories as user history, assessing preference inference and information recall. Operates in a managed open web environment with two tasks.","whyItMatters":"Addresses the gap in web agent benchmarks by capturing personalization from raw browsing history, moving beyond fully explicit prompts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0c774c0c9f6f08299088b528a0909b2f214b1e81776afcf361493bc8a4045b92"},"motivation":"Recent advances in large language models have enabled web agents to autonomously execute complex tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20482","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_perspectivegap_bc1804c7","familyId":"bmf_56adfd1bcc0a","name":"PerspectiveGap","oneLine":"PerspectiveGap is a benchmark with 110 scenarios for evaluating multi-agent orchestration prompting, using distractor-mixed tasks and topologies from the authors' practice.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08878","pdf":"https://arxiv.org/pdf/2606.08878","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.08878"},"evidence":{"snippet":"To measure this, we introduce PerspectiveGap, a benchmark for evaluating LLMs' ability to compose orchestration prompts for multi-agent systems.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08878"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"PerspectiveGap is a benchmark with 110 scenarios for evaluating multi-agent orchestration prompting, using distractor-mixed tasks and topologies from the authors' practice.","whyItMatters":"Multi-agent orchestration prompting is an emerging capability; the benchmark aims to measure it systematically, but the evaluation protocol and artifacts are not publicly accessible.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c015c9f93dddb6a67e3713ed223dd8036fc969e6f1d9b701a49c627692ac36db"},"motivation":"Real-world LLM applications are moving beyond single-agent workflows toward orchestrated multi-agent systems, yet current models still struggle to determine what each sub-agent needs to know.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08878","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_phantombench_848c0022","familyId":"bmf_b3df0dc70b12","name":"PhantomBench","oneLine":"PhantomBench evaluates language models' ability to abstain from answering about non-existent entities. It comprises over 60,000 non-existent terms derived from real concepts across domains, and provides a pipeline for generating further instances.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.11105","pdf":"https://arxiv.org/pdf/2606.11105","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.11105"},"evidence":{"snippet":"We introduce PhantomBench, the first large-scale benchmark of its kind, comprising more than 60K non-existent terms and entities derived from real concepts across diverse domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.11105"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"PhantomBench evaluates language models' ability to abstain from answering about non-existent entities. It comprises over 60,000 non-existent terms derived from real concepts across domains, and provides a pipeline for generating further instances.","whyItMatters":"Addresses the evaluation gap in assessing models' calibration of knowledge boundaries, offering a practical tool for detecting hallucination tendencies in high-stakes applications and studying behavior on rare concepts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2a4e7d3c08cd8f27206db425569fa7bfe03ef54c20ad42a71dfe683da21bace1"},"motivation":"Hallucinations, where language models (LMs) generate factually ungrounded responses, pose serious risks, as users tend to blindly rely on them.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11105","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_phantomfill_4f8c991e","familyId":"bmf_c5647cf4fd51","name":"PhantomFill","oneLine":"PhantomFill measures schema-coerced fabrication in language models by asking questions on unanswerable inputs under three output formats: free text, JSON with an escape option, and JSON with required fields. It reports Coerced Fabrication Rate and Escape Utilization Rate.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20492","pdf":"https://arxiv.org/pdf/2607.20492","project":null,"code":"https://github.com/ranausmanai/phantomfill","data":null,"hfPaper":"https://huggingface.co/papers/2607.20492"},"evidence":{"snippet":"We release PhantomFill, a benchmark with deterministic scoring and two reportable numbers: the Coerced Fabrication Rate and the Escape Utilization Rate.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20492"},"ranking":{"90d":{"score":17,"rank":395,"coverage":0.7,"confidence":"Medium"}},"description":"PhantomFill measures schema-coerced fabrication in language models by asking questions on unanswerable inputs under three output formats: free text, JSON with an escape option, and JSON with required fields. It reports Coerced Fabrication Rate and Escape Utilization Rate.","whyItMatters":"Hallucination in form-filling contexts is under-measured and costly. PhantomFill provides deterministic, code-based metrics targeting a critical failure mode, enabling model comparison and safety evaluation in structured output settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dc0cfa8ee8155946c8493880bd83e29bc0d25bb525a41a8bf1625cd50dde239c"},"motivation":"Language models in production do not write prose.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20492","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_1f366c2487d2dd1f","familyId":"catalog_family_1f366c2487d2dd1f","name":"PhiBench","oneLine":"PhiBench is an internal benchmark designed to evaluate diverse skills and reasoning abilities of language models, covering a wide range of tasks including coding (debugging, extending incomplete code, explaining code snippets) and mathematics (identifying proof errors, generating related problems). Created by Microsoft's research team to address limitations of standard academic benchmarks and guide the development of the Phi-4 model.","description":"PhiBench is an internal benchmark designed to evaluate diverse skills and reasoning abilities of language models, covering a wide range of tasks including coding (debugging, extending incomplete code, explaining code snippets) and mathematics (identifying proof errors, generating related problems). Created by Microsoft's research team to address limitations of standard academic benchmarks and guide the development of the Phi-4 model.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/phibench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1f366c2487d2dd1f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/phibench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"phibench","url":"https://llm-stats.com/benchmarks/phibench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning","general"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_phitsbench_3767e77d","familyId":"bmf_a8ec597bae7e","name":"PHITSBench","oneLine":"PHITSBench evaluates AI-assisted generation of PHITS radiation-transport input via natural language across 282 tasks in three workflows: Edit, Repair, and Reproduce. Scoring uses a Composite Metric Score combining execution success and agreement with reference transport observables.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.09789","pdf":"https://arxiv.org/pdf/2607.09789","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.09789"},"evidence":{"snippet":"We introduce PHITSBench, an execution-scored benchmark for the Monte Carlo Particle and Heavy Ion Transport code System (PHITS).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.09789"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PHITSBench evaluates AI-assisted generation of PHITS radiation-transport input via natural language across 282 tasks in three workflows: Edit, Repair, and Reproduce. Scoring uses a Composite Metric Score combining execution success and agreement with reference transport observables.","whyItMatters":"Provides an execution-grounded benchmark for a niche task, highlighting the need for machine-readable knowledge bases and curated training data in AI-assisted radiation-transport modeling.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"71f35ffb236e6c3b942ee8da7e1ccfc29ca5c8e572bf78082808b3de5eecf18d"},"motivation":"We introduce PHITSBench, an execution-scored benchmark for the Monte Carlo Particle and Heavy Ion Transport code System (PHITS).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.09789","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_phoneharness_125a2bf7","familyId":"bmf_a6d0de088cce","name":"PhoneHarness","oneLine":"PhoneHarness Bench evaluates phone-use agents on verifiable mobile workflows with mixed GUI, CLI, and tool actions, scored by observable side effects from auditable execution traces.","area":"Agents & Tool Use","applicationDomains":["Consumer & Productivity"],"primaryDomain":"Consumer & Productivity","industrySectors":["Consumer Technology"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.14832","pdf":"https://arxiv.org/pdf/2606.14832","project":"https://phoneharness.github.io/","code":"https://github.com/PhoneHarness/PhoneHarness","data":null,"hfPaper":"https://huggingface.co/papers/2606.14832"},"evidence":{"snippet":"We introduce PhoneHarness, a mixed-action benchmark and execution harness for studying phone-use agents on verifiable mobile workflows.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":12,"hfDailySubmittedAt":"2026-06-16T00:00:00.000Z","githubStars":47,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.14832"},"ranking":{"90d":{"score":51,"rank":60,"coverage":0.7,"confidence":"Medium"}},"description":"PhoneHarness Bench evaluates phone-use agents on verifiable mobile workflows with mixed GUI, CLI, and tool actions, scored by observable side effects from auditable execution traces.","whyItMatters":"Mobile agent evaluation often ignores non-GUI actions and side effects; this benchmark measures complete task completion in real device environments, filling a gap for reliable phone automation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"88de4646e887ccf34184dcc526f7bf49218637039c4f01fe9b09cb182befd1cc"},"motivation":"Phone agents are increasingly expected to complete real mobile workflows rather than merely predict the next screen action.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.14832","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_phreeqc-mcq-200_2587cac1","familyId":"bmf_b9702619ebb1","name":"PHREEQC-MCQ-200","oneLine":"PHREEQC-MCQ-200 evaluates tool-augmented agents on 200 multiple-choice questions derived from 21 validated PHREEQC scenarios. Agents must construct simulator inputs, execute PHREEQC, inspect structured outputs, and commit to final answers. Scoring is based on exact match of selected answer.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.00436","pdf":"https://arxiv.org/pdf/2607.00436","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.00436"},"evidence":{"snippet":"We introduce PHREEQC-MCQ-200, a benchmark for evaluating tool-augmented agents on deterministic aqueous-geochemistry simulations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.00436"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PHREEQC-MCQ-200 evaluates tool-augmented agents on 200 multiple-choice questions derived from 21 validated PHREEQC scenarios. Agents must construct simulator inputs, execute PHREEQC, inspect structured outputs, and commit to final answers. Scoring is based on exact match of selected answer.","whyItMatters":"This benchmark addresses the lack of standardized evaluation for tool-augmented agents in scientific simulation, measuring not only accuracy but also item-level retention, output-access sensitivity, and trajectory failures. It provides a diagnostic lens on when tool access improves or degrades performance, informing design of reliable scientific agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e49df8fb53660fcbc0cefac963b5825d9e69d2995ad71adc8252578077b8412f"},"motivation":"Large language model agents are increasingly connected to scientific software, yet it remains unclear when tool access makes scientific computation more reliable rather than merely more complex.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.00436","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"The authors","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.00436","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_phun-bench_8217b7d9","familyId":"bmf_a6eefdc22227","name":"Phun-Bench","oneLine":"Phun-Bench evaluates LLMs' phonological understanding in Chinese across three dimensions: Homophony, Rhyme, and Phonetic Similarity, with diverse tasks designed to isolate genuine phonological ability.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07300","pdf":"https://arxiv.org/pdf/2606.07300","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07300"},"evidence":{"snippet":"Here, we present Phun-Bench, a purpose-built Chinese benchmark with diverse tasks and settings across three dimensions (Homophony, Rhyme, and Phonetic Similarity), designed to systematically evaluate LLMs' phonological understanding.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07300"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Phun-Bench evaluates LLMs' phonological understanding in Chinese across three dimensions: Homophony, Rhyme, and Phonetic Similarity, with diverse tasks designed to isolate genuine phonological ability.","whyItMatters":"This benchmark fills a gap in evaluating phonological abilities beyond semantics and spelling, providing insights into LLMs' flexibility in using sound-based knowledge, which is underexplored.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"069e14924d98da7dc0f03f8cdcecdcbf2af3b8aa5ca6dbef39a9e99c34e39758"},"motivation":"Language is a vehicle for thought, intricately tied to sounds, symbols, and meaning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ACL 2026 Main Conference","evidence":"Accepted to ACL 2026 Main Conference","evidenceUrl":"https://arxiv.org/abs/2606.07300","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ACL 2026 Main Conference","reviewStatus":"accepted","decisionRaw":"Accepted to ACL 2026 Main Conference","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.07300","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to ACL 2026 Main Conference","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_f6b2c1d415c48947","familyId":"catalog_family_f6b2c1d415c48947","name":"PHYBench","oneLine":"PHYBench is a benchmark of real-world physics problems spanning mechanics, electromagnetism, thermodynamics, optics, and modern physics, designed to evaluate physical perception and multi-step quantitative reasoning in large language models.","description":"PHYBench is a benchmark of real-world physics problems spanning mechanics, electromagnetism, thermodynamics, optics, and modern physics, designed to evaluate physical perception and multi-step quantitative reasoning in large language models.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Physics","Reasoning","Science"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/phybench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f6b2c1d415c48947"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/phybench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"phybench","url":"https://llm-stats.com/benchmarks/phybench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["physics","reasoning","science"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_phyeditbench_2e24a281","familyId":"bmf_5eed3359c9a8","name":"PhyEditBench","oneLine":"PhyEditBench evaluates physics-aware image editing. It includes 238 real-world instances and 35 synthetic anti-physics instances across 12 physical subclasses, with VLM-based scoring on consistency, instruction following, physical plausibility, and image quality.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26551","pdf":"https://arxiv.org/pdf/2606.26551","project":null,"code":"https://github.com/Previsior/PhyEditBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.26551"},"evidence":{"snippet":"To address this, we introduce PhyEditBench, a benchmark designed to assess the physical understanding of editing models.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26551"},"ranking":{"90d":{"score":27,"rank":287,"coverage":0.7,"confidence":"Medium"}},"description":"PhyEditBench evaluates physics-aware image editing. It includes 238 real-world instances and 35 synthetic anti-physics instances across 12 physical subclasses, with VLM-based scoring on consistency, instruction following, physical plausibility, and image quality.","whyItMatters":"Image editing benchmarks often overlook physics reasoning; PhyEditBench provides a reusable evaluation to measure physical coherence in edited outputs, important for real-world applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"33de1c095085029be9572bd32768609ba31aa1377a2b22e8cf95778d340dadc1"},"motivation":"While instruction-based image editing, enabled by multi-modal generative models, has advanced significantly, existing benchmarks lack a comprehensive evaluation of physics-based reasoning, a critical capability for handling real-world scenarios.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026","evidence":"19 pages, 6 figures, 2 tables. Accepted to ECCV 2026","evidenceUrl":"https://arxiv.org/abs/2606.26551","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026","reviewStatus":"accepted","decisionRaw":"19 pages, 6 figures, 2 tables. Accepted to ECCV 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.26551","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"19 pages, 6 figures, 2 tables. Accepted to ECCV 2026","level":"author-claim"}]}],"publishers":[{"name":"Previsior","organizationType":"academic-lab","sourceUrl":"https://github.com/Previsior/PhyEditBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_physassistbench_d86f4b50","familyId":"bmf_2c7c02cd01ab","name":"PhysAssistBench","oneLine":"PhysAssistBench evaluates interactive doctor-patient-EHR assistance. It contains 1,296 physician-validated turns from MIMIC-IV cases, testing coordination of clinical knowledge, patient communication, and EHR tool use.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.18613","pdf":"https://arxiv.org/pdf/2606.18613","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18613"},"evidence":{"snippet":"We introduce PhysAssistBench, a benchmark for interactive doctor-patient-EHR assistance.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18613"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PhysAssistBench evaluates interactive doctor-patient-EHR assistance. It contains 1,296 physician-validated turns from MIMIC-IV cases, testing coordination of clinical knowledge, patient communication, and EHR tool use.","whyItMatters":"Current medical LLM evaluations isolate capabilities, but real physician assistance requires integrating knowledge, communication, and systems. PhysAssistBench provides a realistic interaction setting to assess readiness for clinical deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8d021bcb71432ace8c07aa507b4ec7027bf6656593c92cc97484ab5efc69aded"},"motivation":"The most plausible near-term role of medical LLMs is to assist rather than replace physicians, yet current evaluations often test isolated capabilities: clinical knowledge, EHR system interaction, or patient communication.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18613","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_physelite_c8c6e8f9","familyId":"bmf_04959e347f29","name":"PhysElite","oneLine":"Evaluates multimodal LLMs on 11,586 bilingual Olympiad-level physics problems with visual diagrams, step-by-step solutions, and answer accuracy metrics.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-25","firstSeenAt":"2026-08-27","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25097","pdf":"https://arxiv.org/pdf/2608.25097","project":null,"code":null,"data":"https://huggingface.co/datasets/physelite/PhysElite","hfPaper":null},"evidence":{"snippet":"To address these issues, we present PhysElite, a large-scale bilingual multimodal benchmark for Olympiad-level physics reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":67,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2608.25097"},"ranking":{"30d":{"score":49,"rank":32,"coverage":0.15,"confidence":"Low","datasetDownloadRank":24,"datasetRankPopulation":30},"90d":{"score":45,"rank":105,"coverage":0.3,"confidence":"Low","datasetDownloadRank":53,"datasetRankPopulation":66}},"description":"Evaluates multimodal LLMs on 11,586 bilingual Olympiad-level physics problems with visual diagrams, step-by-step solutions, and answer accuracy metrics.","whyItMatters":"Offers a large, expert-level multimodal physics benchmark with step-level process evaluation, addressing gaps in difficulty and visual coverage for physics reasoning.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"fa650160eba61a99a0e484e7819b855c189522cafef4c88f39fe520e71276054"},"motivation":"Understanding how (multimodal) large language models perform on physics problems requires benchmarks that reflect the difficulty and breadth of expert-level physical reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"Dataset is released on Hugging Face, includes answer and step-level scoring, and defines a reusable evaluation protocol for other teams.","canonicalNameSource":"paper_title","canonicalNameEvidence":"PhysElite: How Far Are LLMs from Solving Olympiad-Level Physics Problems?"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25097","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":72,"confidence":"Medium","horizon":"7d","reason":"Large-scale multilingual benchmark tackling challenging physics reasoning is likely to draw broad interest from the multimodal and scientific AI communities."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception","Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_physicsbench_c13eff48","familyId":"bmf_b34dce60cd36","name":"PhysicsBench","oneLine":"PhysicsBench is a unified benchmark and leaderboard for generative and predictive AI models in engineering design and simulation. It spans seven tasks across 1D, 2D, and 3D domains, ranks 66 models on nine datasets, and uses a common metric suite including BenchRank for debiased ranking. Evaluation covers data scales from S to XL.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-25","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.24056","pdf":"https://arxiv.org/pdf/2608.24056","project":"https://leaderboard.narnia.ai","code":"https://github.com/Narnialabs/leaderboard","data":null,"hfPaper":null},"evidence":{"snippet":"We present PhysicsBench, a unified benchmark and leaderboard that evaluates generative and predictive models under one standardized procedure.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24056"},"ranking":{"30d":{"score":33,"rank":106,"coverage":0.55,"confidence":"Low"},"90d":{"score":29,"rank":320,"coverage":0.55,"confidence":"Low"},"today":{"score":56,"rank":1,"coverage":0.25,"confidence":"Low"}},"description":"PhysicsBench is a unified benchmark and leaderboard for generative and predictive AI models in engineering design and simulation. It spans seven tasks across 1D, 2D, and 3D domains, ranks 66 models on nine datasets, and uses a common metric suite including BenchRank for debiased ranking. Evaluation covers data scales from S to XL.","whyItMatters":"Existing evaluations of generative and predictive AI in engineering are isolated with inconsistent metrics. PhysicsBench provides a standardized leaderboard for model selection, revealing that academic performance poorly predicts small-data ranking, supporting data-efficiency decisions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"13a6a5fd5b74275848082aa511047279b8707786f76252cc1bbf8f25a1d922ef"},"motivation":"Generative and predictive artificial intelligence models are increasingly used to generate geometry and to predict physical fields and scalar quantities in engineering design and simulation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark has an official project page, live leaderboard, and code repository, indicating a public reuse path and stable scoring contract.","canonicalNameSource":"abstract","canonicalNameEvidence":"We present PhysicsBench, a unified benchmark and leaderboard"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24056","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":80,"confidence":"Medium","horizon":"7d","reason":"Practical engineering relevance and a live leaderboard with broad model coverage are likely to draw industrial and academic attention."},"evaluationMode":"score_submission","publishers":[{"name":"Narnia Labs","organizationType":"company-research-lab","sourceUrl":"https://leaderboard.narnia.ai","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_dc3fcf874da57885","familyId":"catalog_family_dc3fcf874da57885","name":"PhysicsFinals","oneLine":"PHYSICS is a comprehensive benchmark for university-level physics problem solving, containing 1,297 expert-annotated problems covering six core areas: classical mechanics, quantum mechanics, thermodynamics and statistical mechanics, electromagnetism, atomic physics, and optics. Each problem requires advanced physics knowledge and mathematical reasoning. Even advanced models like o3-mini achieve only 59.9% accuracy.","description":"PHYSICS is a comprehensive benchmark for university-level physics problem solving, containing 1,297 expert-annotated problems covering six core areas: classical mechanics, quantum mechanics, thermodynamics and statistical mechanics, electromagnetism, atomic physics, and optics. Each problem requires advanced physics knowledge and mathematical reasoning. Even advanced models like o3-mini achieve only 59.9% accuracy.","area":"Mathematical Reasoning","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Math","Physics","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/physicsfinals","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dc3fcf874da57885"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/physicsfinals"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"physicsfinals","url":"https://llm-stats.com/benchmarks/physicsfinals","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","physics","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"bm_phystool-bench_e6d3d9ac","familyId":"bmf_2635f961d28a","name":"PhysTool-Bench","oneLine":"PhysTool-Bench evaluates multimodal LLMs on physical tool use through two tasks: recognizing all tools in a scene and selecting and sequencing tools for a given task, using 2,510 queries over 2,678 real-world tools.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.10803","pdf":"https://arxiv.org/pdf/2606.10803","project":null,"code":"https://github.com/ModalityDance/PhysTool-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.10803"},"evidence":{"snippet":"To address this gap, we introduce PhysTool-Bench, the first physical tool-use benchmark designed to evaluate MLLMs' ability to comprehend real-world scenarios, identify physical tools, and plan their use.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.10803"},"ranking":{"90d":{"score":19,"rank":391,"coverage":0.7,"confidence":"Medium"}},"description":"PhysTool-Bench evaluates multimodal LLMs on physical tool use through two tasks: recognizing all tools in a scene and selecting and sequencing tools for a given task, using 2,510 queries over 2,678 real-world tools.","whyItMatters":"Physical tool use is underexplored in MLLMs; this benchmark isolates recognition and planning deficits, supporting progress in embodied AI and practical human-robot collaboration.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a6a97b05aa973e34994b7af031a29d2daf8f7f3edeaaf56c4d8b92a3b7adf350"},"motivation":"Multimodal Large Language Models (MLLMs) excel at utilizing digital APIs and increasingly serve as the \"brain\" of embodied AI, instructing robots to interact with the physical world.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.10803","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ModalityDance","organizationType":"academic-lab","sourceUrl":"https://github.com/ModalityDance/PhysTool-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_c7d25132dcc30c2c","familyId":"catalog_family_c7d25132dcc30c2c","name":"PinchBench","oneLine":"PinchBench evaluates coding agents on real-world agentic coding tasks, measuring both best-case and average performance across complex software engineering scenarios.","description":"PinchBench evaluates coding agents on real-world agentic coding tasks, measuring both best-case and average performance across complex software engineering scenarios.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://pinchbench.com/about","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c7d25132dcc30c2c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/pinchbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/pinchbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"pinchBench","url":"https://benchlm.ai/benchmarks/pinchbench","paperUrl":"https://pinchbench.com/about","year":"2026","fullName":"PinchBench","format":"Average success rate from official runs","tasks":"23 OpenClaw agent tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"pinchbench","url":"https://llm-stats.com/benchmarks/pinchbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","agents","code"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_pinverify_ff022fa6","familyId":"bmf_220fd0a401a3","name":"PInVerify","oneLine":"PInVerify is an offline embodied benchmark for Active Instance Verification, where agents select viewpoints around a candidate object to decide if it matches a fine-grained description.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2605.30639","pdf":"https://arxiv.org/pdf/2605.30639","project":null,"code":"https://github.com/Avalon-S/PInVerify","data":null,"hfPaper":"https://huggingface.co/papers/2605.30639"},"evidence":{"snippet":"We formalize AIV as a finite-horizon decision process and introduce PInVerify, an offline embodied benchmark for AIV: 3,000 evaluation episodes across 18 object categories, delivered as multi-view captures with a 6-sector navigation topology that exposes trap views (navigable but uninformative) and unreachable sectors.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30639"},"ranking":{},"description":"PInVerify is an offline embodied benchmark for Active Instance Verification, where agents select viewpoints around a candidate object to decide if it matches a fine-grained description.","whyItMatters":"It addresses the gap between navigating to an object and verifying its identity through active perception, which is crucial for embodied agents in fine-grained tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b6c2b78eccc22405cf4845b1bdde33171efb29a708a94becb0998c316ebd319d"},"motivation":"Embodied agents have made strong progress in navigating to target objects, but reaching the goal vicinity does not guarantee that the agent has found the correct instance: subtle attribute differences (e.g., \"white floral\" vs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"as a poster at the Foundation Models Meet Embodied Agents (FMEA) Workshop, CVPR 2026","evidence":"Accepted as a poster at the Foundation Models Meet Embodied Agents (FMEA) Workshop, CVPR 2026. 44 pages including appendix. Code: https://github.com/Avalon-S/PInVerify","evidenceUrl":"https://arxiv.org/abs/2605.30639","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"as a poster at the Foundation Models Meet Embodied Agents (FMEA) Workshop, CVPR 2026","reviewStatus":"accepted","decisionRaw":"Accepted as a poster at the Foundation Models Meet Embodied Agents (FMEA) Workshop, CVPR 2026. 44 pages including appendix. Code: https://github.com/Avalon-S/PInVerify","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2605.30639","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted as a poster at the Foundation Models Meet Embodied Agents (FMEA) Workshop, CVPR 2026. 44 pages including appendix. Code: https://github.com/Avalon-S/PInVerify","level":"author-claim"}]}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_pipbench_0f193694","familyId":"bmf_76865cb5cf7c","name":"PIPBench","oneLine":"PIPBench evaluates personalized image generation, where models must align outputs with a user's implicit visual preferences based on a few historically preferred images and a short prompt. It includes real-user and agent-based data across psychological and demographic profiles.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.06440","pdf":"https://arxiv.org/pdf/2607.06440","project":"https://wuyuhang05.github.io/PIPBench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.06440"},"evidence":{"snippet":"To this end, we introduce PIPBench, the first profile-inclusive benchmark for evaluating personalized image generation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06440"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"PIPBench evaluates personalized image generation, where models must align outputs with a user's implicit visual preferences based on a few historically preferred images and a short prompt. It includes real-user and agent-based data across psychological and demographic profiles.","whyItMatters":"Existing text-to-image benchmarks focus on prompt following but ignore individual aesthetic preferences. PIPBench addresses the evaluation gap for personalized generation, offering a standardized way to compare methods aligning outputs with user profiles.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7c3605029971c551431e90afe5d31d46d04c2f5a5e1b1d5c900b143416d8891a"},"motivation":"Recent text-to-image models such as DALLE-3 excel at following diverse prompts yet remain blind to individual aesthetic preferences.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06440","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"PIPBench Team","organizationType":"academic-lab","sourceUrl":"https://wuyuhang05.github.io/PIPBench/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_pipe-cypher_c078ae44","familyId":"bmf_2fdb7077d9b5","name":"PIPE-Cypher","oneLine":"PIPE-Cypher is a pipeline that generates NL-to-Cypher benchmarks from live property graphs, producing executable query pairs with validation, diversity controls, and local LLM judges.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08481","pdf":"https://arxiv.org/pdf/2606.08481","project":null,"code":"https://github.com/suraj-ranganath/PIPE-Cypher","data":null,"hfPaper":"https://huggingface.co/papers/2606.08481"},"evidence":{"snippet":"We present PIPE-Cypher, a local benchmark-generation pipeline that turns a live property graph and optional seed queries from customer questions, analyst logs, or agent tool calls into balanced NL-to-Cypher benchmarks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":"2026-06-09T00:00:00.000Z","githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08481"},"ranking":{"90d":{"score":15,"rank":405,"coverage":0.7,"confidence":"Medium"}},"description":"PIPE-Cypher is a pipeline that generates NL-to-Cypher benchmarks from live property graphs, producing executable query pairs with validation, diversity controls, and local LLM judges.","whyItMatters":"Enterprise graph schemas and query patterns are unique and evolve, making static benchmarks obsolete. PIPE-Cypher enables repeatable, graph-specific evaluation of text-to-Cypher systems, supporting deployment-relevant performance assessment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e302ac4fea949d587350ca60e3b062e2c020656546bab6ccf270966c4a4cd33b"},"motivation":"Enterprise property graphs vary widely in schema structure, internal terminology, domain assumptions, governance constraints, and user interaction patterns.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08481","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"PIPE-Cypher Developers","organizationType":"community","sourceUrl":"https://github.com/suraj-ranganath/PIPE-Cypher","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_7f8d07159d2aa133","familyId":"catalog_family_7f8d07159d2aa133","name":"PIQA","oneLine":"PIQA (Physical Interaction: Question Answering) is a benchmark dataset for physical commonsense reasoning in natural language. It tests AI systems' ability to answer questions requiring physical world knowledge through multiple choice questions with everyday situations, focusing on atypical solutions inspired by instructables.com. The dataset contains 21,000 multiple choice questions where models must choose the most appropriate solution for physical interactions.","description":"PIQA (Physical Interaction: Question Answering) is a benchmark dataset for physical commonsense reasoning in natural language. It tests AI systems' ability to answer questions requiring physical world knowledge through multiple choice questions with everyday situations, focusing on atypical solutions inspired by instructables.com. The dataset contains 21,000 multiple choice questions where models must choose the most appropriate solution for physical interactions.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Physics","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/piqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7f8d07159d2aa133"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/piqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"piqa","url":"https://llm-stats.com/benchmarks/piqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["physics","reasoning","general"],"catalogModelCount":11,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_pitchbench_7fffde93","familyId":"bmf_ccf6122c0774","name":"PitchBench","oneLine":"PitchBench evaluates pitch hearing in audio-language models across 28 experiments spanning absolute and relative pitch perception in sequences and chords, varying acoustic conditions and response formats.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SD"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26176","pdf":"https://arxiv.org/pdf/2605.26176","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.26176"},"evidence":{"snippet":"We introduce PitchBench, an evaluation suite that systematically measures pitch hearing in ALMs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26176"},"ranking":{},"description":"PitchBench evaluates pitch hearing in audio-language models across 28 experiments spanning absolute and relative pitch perception in sequences and chords, varying acoustic conditions and response formats.","whyItMatters":"Pitch perception is foundational for musical reasoning, yet existing benchmarks probe it indirectly. PitchBench provides a systematic, controlled evaluation to identify limitations in current models and support the development of pitch-aware audio-language systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3adb7e98ef93ba2631d3d5782ce14338969a9f02f423d508df118dece5bc5f18"},"motivation":"Audio-language models (ALMs) are increasingly used in real-world applications that require understanding music, from music tutoring and transcription to captioning, recommendation systems, and music production.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26176","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_pivot-a-multi-trajectory-dataset-and-testb_3fccb501","familyId":"bmf_cebf0aa15ec2","name":"PIVOT","oneLine":"Evaluates novel-view synthesis under diverse camera trajectories, measured vs optimized poses, and calibrated vs optimized intrinsics using five real-world scenes.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-26","firstSeenAt":"2026-08-27","recognitionConfidence":0.45,"links":{"report":"https://arxiv.org/abs/2608.25401","pdf":"https://arxiv.org/pdf/2608.25401","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce PIVOT (Pose, Intrinsics and Viewpoint Oriented Testbed), a multi-trajectory dataset, processing pipeline, and evaluation framework for independently studying these factors.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25401"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates novel-view synthesis under diverse camera trajectories, measured vs optimized poses, and calibrated vs optimized intrinsics using five real-world scenes.","whyItMatters":"Reveals performance gaps in reconstruction methods under conditions closer to robotic deployment than standard novel-view benchmarks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"8868b62f8eb2cb02f116d59705cd4ffd28576c8bd9f242d53d525b00c6860271"},"motivation":"Neural radiance fields (NeRFs), 3D Gaussian Splatting (3DGS), and related novel-view synthesis methods are commonly evaluated under capture and reconstruction conditions cleaner than those encountered by robots, drones, and autonomous systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:12:36.511123Z","model":"deepseek-v4-pro","decisionReason":"Abstract declares an open processing and evaluation toolchain plus benchmark families with clear scoring dimensions; PIVOT is explicitly named and defined as a testbed.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce PIVOT (Pose, Intrinsics and Viewpoint Oriented Testbed), a multi-trajectory dataset, processing pipeline, and evaluation framework"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25401","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":55,"confidence":"Low","horizon":"7d","reason":"Topical 3D reconstruction benchmark with an open toolchain, though limited to five scenes and academic visibility."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_pivotsbench_17265409","familyId":"bmf_c9a766b802d9","name":"PIVOTSBench","oneLine":"PIVOTSBench evaluates multimodal large language models on fine-grained interpersonal relationship reasoning. It includes tasks predicting bidirectional relationship dimensions from videos and auxiliary tasks on visual cue identification.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.23092","pdf":"https://arxiv.org/pdf/2606.23092","project":"https://flynnzhangsx.github.io/PIVOTSBench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.23092"},"evidence":{"snippet":"To address this gap, we introduce PIVOTS, the first benchmark built from Social-IQ 2.0 and YouTube data to evaluate MLLMs' ability to predict bidirectional interpersonal relationship dimensions grounded in established psychology research.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.23092"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PIVOTSBench evaluates multimodal large language models on fine-grained interpersonal relationship reasoning. It includes tasks predicting bidirectional relationship dimensions from videos and auxiliary tasks on visual cue identification.","whyItMatters":"This benchmark addresses the lack of evaluation for multimodal social reasoning, providing a standardized test for model capabilities in understanding nuanced interpersonal cues, which is crucial for developing AI that interacts naturally in social contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4ad2fc3ec8eb2794d1970a9432d771d0c35c8415d44e9c90878ad0b7870409f8"},"motivation":"Humans possess an innate ability to understand fine-grained interpersonal relationships, which is central to everyday social interactions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.23092","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_pl-nba_96aca2fb","familyId":"bmf_945d8d97150d","name":"PL-NBA","oneLine":"Dataset of 11,000 NBA possession clips with 31,567 annotated events for basketball video understanding tasks like event recognition, captioning, localization, and anticipation.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-21","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.19646","pdf":"https://arxiv.org/pdf/2608.19646","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Experimental results show that existing methods achieve limited performance on above four tasks, demonstrating that PL-NBA is a challenging benchmark for sports video understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.19646"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Dataset of 11,000 NBA possession clips with 31,567 annotated events for basketball video understanding tasks like event recognition, captioning, localization, and anticipation.","whyItMatters":"Introduces a possession-level dataset that preserves temporal continuity, enabling evaluation across multiple video understanding tasks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"0f7f220faaac2da3647b8f74c2b35037fe628e507f93ab385e1c22bb93e4241b"},"motivation":"Visual understanding in sports has emerged as a hot topic in computer vision in recent years.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The paper presents a new dataset with defined tasks, but no public artifacts are yet linked, limiting reusability for now."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.19646","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-21T04:30:09.569171Z"},"attentionForecast":{"score":40,"confidence":"Low","horizon":"7d","reason":"Sports video understanding is a moderately active area, but lack of public data may reduce immediate attention."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_plan2map_b8494f1d","familyId":"bmf_133ef96ac8ba","name":"Plan2Map","oneLine":"Plan2Map evaluates document-grounded geospatial boundary reconstruction from UK planning records. Systems input a planning document and output a GeoJSON boundary, scored against held-out reference boundaries via IoU.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02747","pdf":"https://arxiv.org/pdf/2606.02747","project":"https://odeb1.github.io/Plan2Map_Project_Page/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02747"},"evidence":{"snippet":"We introduce Plan2Map, a 208-case multimodal benchmark for document-grounded geospatial boundary reconstruction from UK planning records.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02747"},"ranking":{},"description":"Plan2Map evaluates document-grounded geospatial boundary reconstruction from UK planning records. Systems input a planning document and output a GeoJSON boundary, scored against held-out reference boundaries via IoU.","whyItMatters":"Plan2Map addresses the gap in evaluating multimodal geospatial reconstruction from public planning documents, providing a concrete testbed with held-out scoring for comparing systems on a complex, real-world task.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"09e8bc7061dbfe64ade1b66c56382c2c095ae5aff0cb41b70498d2bd7946357c"},"motivation":"Planning records define restrictions over geographic areas, but their source documents often provide only indirect spatial evidence rather than machine-readable boundaries.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02747","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_planbench-v_89620860","familyId":"bmf_88e93d3be47e","name":"PlanBench-V","oneLine":"PlanBench-V evaluates vision-language models on spatial planning map interpretation through an expert-annotated dataset of 223 maps and 1629 question-answer pairs, assessing perception, reasoning, association, and implementation capabilities.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05744","pdf":"https://arxiv.org/pdf/2606.05744","project":"https://plangpt.github.io","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05744"},"evidence":{"snippet":"To address this gap, we introduce PlanBench-V, the first comprehensive benchmark for evaluating VLMs in spatial planning map interpretation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05744"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PlanBench-V evaluates vision-language models on spatial planning map interpretation through an expert-annotated dataset of 223 maps and 1629 question-answer pairs, assessing perception, reasoning, association, and implementation capabilities.","whyItMatters":"Existing multimodal benchmarks overlook domain-specific spatial planning tasks. PlanBench-V provides a theory-informed framework for evaluating VLM progress in professional planning contexts, identifying persistent limitations in implementation-oriented tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3070dd12d830bab4e08ddd939d1b2d731af50c9816dcda640c3220c66362a9e2"},"motivation":"Spatial planning maps are central to territorial governance, translating planning objectives, regulations, and spatial strategies into visual forms for decision-making, public communication, and institutional coordination.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05744","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"PlanGPT","organizationType":"academic-lab","sourceUrl":"https://plangpt.github.io","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_planbench-xl_459bd32f","familyId":"bmf_023188610684","name":"PlanBench-XL","oneLine":"PlanBench-XL evaluates LLM tool-use agents on long-horizon planning in large-scale tool ecosystems. It includes 327 retail tasks across 1,665 tools, requiring iterative tool retrieval and use. Optional blocker mechanisms inject missing, failing, or distracting tools to test adaptive planning. Scoring is based on final answer accuracy and auxiliary metrics.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22388","pdf":"https://arxiv.org/pdf/2606.22388","project":null,"code":"https://github.com/JiayuJeff/PlanBench-XL","data":null,"hfPaper":"https://huggingface.co/papers/2606.22388"},"evidence":{"snippet":"To address this gap, we introduce PlanBench-XL, an interactive benchmark of 327 retail tasks over 1,665 tools that tests whether agents can iteratively retrieve usable tools, invoke them to uncover intermediate evidence for subsequent calls toward the final goal.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":96,"hfDailySubmittedAt":"2026-06-23T00:00:00.000Z","githubStars":40,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22388"},"ranking":{"90d":{"score":55,"rank":35,"coverage":0.7,"confidence":"Medium"}},"description":"PlanBench-XL evaluates LLM tool-use agents on long-horizon planning in large-scale tool ecosystems. It includes 327 retail tasks across 1,665 tools, requiring iterative tool retrieval and use. Optional blocker mechanisms inject missing, failing, or distracting tools to test adaptive planning. Scoring is based on final answer accuracy and auxiliary metrics.","whyItMatters":"Existing benchmarks often assume full tool visibility, which underrepresents real-world agent deployment. PlanBench-XL fills this gap by testing planning under retrieval-limited visibility and tool failures, providing insight into robustness and adaptability of LLM agents in complex tool environments.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cbfd3b537693765f431621cd94544bd7036ffce327d95eff0c803eff393bfa90"},"motivation":"LLM agents increasingly operate in large tool ecosystems, where real-world tasks require discovering relevant tools, inferring implicit sub-goals, and adapting to dynamic environments over long horizons.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22388","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"PlanBench-XL Team","organizationType":"academic-lab","sourceUrl":"https://github.com/JiayuJeff/PlanBench-XL","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_77674378a36a0f93","familyId":"catalog_family_77674378a36a0f93","name":"PLawBench","oneLine":"PLawBench evaluates language models on professional legal knowledge and reasoning tasks.","description":"PLawBench evaluates language models on professional legal knowledge and reasoning tasks.","area":"Language & Knowledge","applicationDomains":["Law & Government"],"primaryDomain":"Law & Government","industrySectors":[],"capabilities":[],"topics":["Knowledge","Legal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/plawbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_77674378a36a0f93"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/plawbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"plawbench","url":"https://llm-stats.com/benchmarks/plawbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","legal","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_playworld_54fed7c4","familyId":"bmf_dc419bdc19b6","name":"PlayWorld","oneLine":"PlayWorld evaluates interactive video world models across 171 scenarios with long-horizon objectives. Agent players adaptively control each model, and performance is scored on geometry consistency, interaction fidelity, out-of-sight evolution, and insight evolution, plus basic video quality and controllability metrics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.13552","pdf":"https://arxiv.org/pdf/2608.13552","project":"https://kxding.github.io/project/PlayWorld/","code":"https://github.com/kxding/PlayWorld","data":null,"hfPaper":"https://huggingface.co/papers/2608.13552"},"evidence":{"snippet":"Building on this paradigm, we introduce PlayWorld, a benchmark providing 171 scenarios, each with a specified objective.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":46,"hfDailySubmittedAt":"2026-08-14T00:00:00.000Z","githubStars":87,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13552"},"ranking":{"30d":{"score":65,"rank":6,"coverage":0.85,"confidence":"High"},"90d":{"score":59,"rank":24,"coverage":0.7,"confidence":"Medium"}},"description":"PlayWorld evaluates interactive video world models across 171 scenarios with long-horizon objectives. Agent players adaptively control each model, and performance is scored on geometry consistency, interaction fidelity, out-of-sight evolution, and insight evolution, plus basic video quality and controllability metrics.","whyItMatters":"Standard fixed-action evaluations fail to compare world models that handle actions differently. PlayWorld provides an objective-driven protocol that reflects real user interaction, enabling fair cross-model assessment and highlighting limitations in spatial consistency and state evolution for long-horizon tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8ec6430d7c17387d79353c7e4f6c3b55dd28417365f2b8dd3d5533c88d917fe9"},"motivation":"Video world models simulate future states conditioned on current observations and user actions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13552","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_plcbench_fb7005e5","familyId":"bmf_db5390e10777","name":"PLCBENCH","oneLine":"A hardware-in-the-loop framework evaluating LLM agents across commercial PLCs and physical process simulations, with deterministic scoring for PLC interaction, process-linked manipulation, and sustained physical impact.","area":"Agents & Tool Use","applicationDomains":["Industrial & Engineering","Robotics & Autonomous Systems"],"primaryDomain":"Industrial & Engineering","industrySectors":["Manufacturing","Robotics"],"capabilities":["Factuality"],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":0.65,"links":{"report":"https://arxiv.org/abs/2608.26882","pdf":"https://arxiv.org/pdf/2608.26882","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present PLCBENCH, to our knowledge, the first real-PLC hardware-in-the-loop (HIL) framework for characterizing this cyber-to-physical capability and its boundaries.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":null,"hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26882"},"ranking":{},"description":"A hardware-in-the-loop framework evaluating LLM agents across commercial PLCs and physical process simulations, with deterministic scoring for PLC interaction, process-linked manipulation, and sustained physical impact.","whyItMatters":"Existing evaluations stop before measuring sustained physical impact, leaving cyber-physical risk unquantified for autonomous agents targeting ICS environments.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T07:23:36.749451Z","inputHash":"02d6e53d0aebf0b5266d7e2f249cf1923e91d6da0d522a9695d8187add13d5bd"},"motivation":"Industrial control systems (ICSs) rely on programmable logic controllers (PLCs) to connect networked computation with physical control.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T07:23:36.749451Z","model":"deepseek-v4-pro","decisionReason":"Declared benchmark with released code, software-only pipeline, deterministic evaluator, and reusable hardware/software setup; does not emphasize ongoing submission leaderboard but supports independent reproduction.","canonicalNameSource":"abstract","canonicalNameEvidence":"We present PLCBENCH, to our knowledge, the first real-PLC hardware-in-the-loop (HIL) framework"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.26882","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-30T06:59:55.242506Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"Novel real-world cyber-physical agent evaluation with strong release claims likely attracts security researchers and practitioners within the first week."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Agents"],"domainScope":"cross-domain"},{"id":"bm_plsqlbench_adc42146","familyId":"bmf_5ba5b1f87407","name":"PLSQLBench","oneLine":"PLSQLBench evaluates LLMs' ability to write executable PL/SQL programs through execution-based tests. It contains 2,865 instances including single-turn and multi-turn tasks, covering schema-grounded and procedural problems.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation","Factuality"],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15931","pdf":"https://arxiv.org/pdf/2608.15931","project":null,"code":"https://github.com/oracle-samples/plsqlbench","data":null,"hfPaper":"https://huggingface.co/papers/2608.15931"},"evidence":{"snippet":"We present PLSQLBench, to our knowledge the first benchmark for evaluating whether LLMs can write executable PL/SQL programs, with correctness measured through execution-based tests.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15931"},"ranking":{"30d":{"score":23,"rank":150,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":354,"coverage":0.55,"confidence":"Low"}},"description":"PLSQLBench evaluates LLMs' ability to write executable PL/SQL programs through execution-based tests. It contains 2,865 instances including single-turn and multi-turn tasks, covering schema-grounded and procedural problems.","whyItMatters":"Existing evaluations target general code generation or declarative text-to-SQL, leaving procedural database programming underexplored. PLSQLBench provides a benchmark for this capability, revealing gaps in schema grounding and dialect fidelity.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4e3af412c98d67b6c8c4c851ca509ffe6046362ee48cb7631a961e15fed3543f"},"motivation":"We present PLSQLBench, to our knowledge the first benchmark for evaluating whether LLMs can write executable PL/SQL programs, with correctness measured through execution-based tests.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15931","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Oracle Samples","organizationType":"company-research-lab","sourceUrl":"https://github.com/oracle-samples/plsqlbench","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_plugineval_1642b04d","familyId":"bmf_604048e8cd8d","name":"PluginEval","oneLine":"PluginEval evaluates tool routing in LLMs via three decision types (missed, spurious, parameter errors) across difficulty levels, using deterministic validation and real API execution for reliable signals.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.08700","pdf":"https://arxiv.org/pdf/2608.08700","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.08700"},"evidence":{"snippet":"In this paper, we introduce PluginEval, a benchmark constructed through a two-stage framework that systematically mitigates these limitations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08700"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PluginEval evaluates tool routing in LLMs via three decision types (missed, spurious, parameter errors) across difficulty levels, using deterministic validation and real API execution for reliable signals.","whyItMatters":"It overcomes limitations of power-law data distributions and unvalidated LLM judgments, offering a diagnostic error profile for function calling and enabling more reliable agent evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e7eaa41ec8defa91b5c9ad920ac29c0c8093ad0dd1c09846df78e9190ed9dbc7"},"motivation":"Reliable evaluation of tool routing is critical as Large Language Models increasingly operate as autonomous agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.08700","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_pm-bench_3de97f77","familyId":"bmf_ecbb460ae2fa","name":"PM-Bench","oneLine":"PM-Bench evaluates prospective memory in LLM agents through a text-based simulated seven-day week. Agents must maintain user intentions, execute delayed intentions, and monitor latent environment changes while performing ongoing activities.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.12385","pdf":"https://arxiv.org/pdf/2607.12385","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.12385"},"evidence":{"snippet":"We introduce PM-Bench, a text-based benchmark for measuring prospective memory capabilities in modern LLM agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.12385"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PM-Bench evaluates prospective memory in LLM agents through a text-based simulated seven-day week. Agents must maintain user intentions, execute delayed intentions, and monitor latent environment changes while performing ongoing activities.","whyItMatters":"PM-Bench fills a gap in evaluating agentic AI by measuring an understudied cognitive capability—prospective memory—in a controlled, repeatable setting. It provides a diagnostic tool for comparing LLM agents and guiding interventions to improve reliability in real-world tasks requiring memory for future actions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"57ee1bd17b05250289c0cbe599796abada7eb96ca182b1109ed88ec2a995a9e2"},"motivation":"A significant challenge in agentic AI is prospective memory: the ability to execute an intention at a specific future cue or state while other activities are ongoing.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.12385","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_be4b2f868732f6bc","familyId":"catalog_family_be4b2f868732f6bc","name":"PMC-VQA","oneLine":"A medical visual question answering benchmark built on biomedical literature and medical figures.","description":"A medical visual question answering benchmark built on biomedical literature and medical figures.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Healthcare","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/pmc-vqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_be4b2f868732f6bc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/pmc-vqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"pmc-vqa","url":"https://llm-stats.com/benchmarks/pmc-vqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","healthcare","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_pocket-dentist_999be450","familyId":"bmf_da4910d02d89","name":"Pocket-Dentist","oneLine":"Pocket-Dentist is an efficiency-aware benchmark for dental multimodal question answering that combines three datasets, five task types, and seven metrics to evaluate VLMs on accuracy and computational cost.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29299","pdf":"https://arxiv.org/pdf/2605.29299","project":"https://2026-icml.github.io/pocket-dentist-icml","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29299"},"evidence":{"snippet":"Here we present Pocket-Dentist, an efficiency-aware benchmark for dental multimodal question answering that brings together three datasets spanning approximately 1,159 patients from BRAR and MetaDent, five task types and seven metrics.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29299"},"ranking":{},"description":"Pocket-Dentist is an efficiency-aware benchmark for dental multimodal question answering that combines three datasets, five task types, and seven metrics to evaluate VLMs on accuracy and computational cost.","whyItMatters":"It highlights the trade-off between model performance and efficiency for on-device dental screening, which is essential for practical deployment in resource-limited settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f37b4a6cc7843f5b88f9ce91d57160bb26c3ea58bcf550d9c625eb39e946b835"},"motivation":"Evaluations of dental vision-language models remain fragmented across datasets, task definitions and metrics, and often ignore their computational cost.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29299","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_poinav-bench_7e6a4ee3","familyId":"bmf_aca8b20281af","name":"POINav-Bench","oneLine":"POINav-Bench evaluates vision-language navigation agents in real-world POI-goal navigation across 11 reconstructed commercial areas covering 126,398 m² with 163 POIs, using traversability-aware annotations and reference trajectories for closed-loop evaluation.","area":"Robotics & Embodied AI","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.28237","pdf":"https://arxiv.org/pdf/2605.28237","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28237"},"evidence":{"snippet":"To bridge this gap, we present POINav-Bench, the first benchmark designed for closed-loop evaluation of real-world POI-goal navigation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28237"},"ranking":{},"description":"POINav-Bench evaluates vision-language navigation agents in real-world POI-goal navigation across 11 reconstructed commercial areas covering 126,398 m² with 163 POIs, using traversability-aware annotations and reference trajectories for closed-loop evaluation.","whyItMatters":"Existing VLN benchmarks for POI-goal navigation suffer from coarse granularity or sim-to-real gaps. POINav-Bench provides high-fidelity real-world environments, enabling evaluation of final-meters navigation capabilities that are critical for practical deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"a765471fff8234d74b41ab3f4bab81b320d84dfbdde8b5f4da1a57fbabb8a576"},"motivation":"Real-world navigation is fundamentally driven by Points of Interest (POIs), yet reaching a precise POI remains a critical \"final-meters\" challenge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28237","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"POINav Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2605.28237","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"general"},{"id":"catalog_9b11c51afdac427a","familyId":"catalog_family_9b11c51afdac427a","name":"PointGrounding","oneLine":"PointArena is a comprehensive platform for evaluating multimodal pointing across diverse reasoning scenarios. It includes Point-Bench, a curated dataset of ~1,000 pointing tasks across five categories: Spatial (positional references), Affordance (functional part identification), Counting (attribute-based grouping), Steerable (relative pointing), and Reasoning (open-ended visual inference). The benchmark evaluates language-guided pointing capabilities in vision-language models.","description":"PointArena is a comprehensive platform for evaluating multimodal pointing across diverse reasoning scenarios. It includes Point-Bench, a curated dataset of ~1,000 pointing tasks across five categories: Spatial (positional references), Affordance (functional part identification), Counting (attribute-based grouping), Steerable (relative pointing), and Reasoning (open-ended visual inference). The benchmark evaluates language-guided pointing capabilities in vision-language models.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Spatial Reasoning","Grounding","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/pointgrounding","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9b11c51afdac427a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/pointgrounding"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"pointgrounding","url":"https://llm-stats.com/benchmarks/pointgrounding","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","spatial reasoning","grounding","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_pointq-bench_51211dca","familyId":"bmf_6e95a65fb614","name":"PointQ-Bench","oneLine":"PointQ-Bench is a benchmark for point cloud quality assessment, extending from scalar scoring to comprehensive quality understanding, with 3,083 point clouds and tasks like anomaly sensing, defect diagnosis, usability grading, and open-ended quality reporting.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.28241","pdf":"https://arxiv.org/pdf/2605.28241","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28241"},"evidence":{"snippet":"We introduce PointQ-Bench, a benchmark designed to extend PCQA from scalar scoring toward comprehensive quality understanding.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28241"},"ranking":{},"description":"PointQ-Bench is a benchmark for point cloud quality assessment, extending from scalar scoring to comprehensive quality understanding, with 3,083 point clouds and tasks like anomaly sensing, defect diagnosis, usability grading, and open-ended quality reporting.","whyItMatters":"Current PCQA benchmarks focus on scalar prediction, leaving gaps in diagnostic and interpretable quality assessment. PointQ-Bench addresses this by evaluating models on multi-faceted quality understanding tasks, which is crucial for practical inspection scenarios where identifying defects and assessing usability is necessary.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7806450d88448de6f425cb752c2577b82256e1ca77d0b2d57f39cd1d2d5bf2be"},"motivation":"Point cloud quality plays a critical role in 3D acquisition, reconstruction, rendering, and perception, yet existing point cloud quality assessment (PCQA) research remains largely centered on scalar score prediction.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28241","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_poisonforge_b7c26795","familyId":"bmf_d4f13d62e72d","name":"PoisonForge","oneLine":"PoisonForge benchmarks task-level targeted poisoning of instruction-tuned LLMs, parameterizing bias type, poisoning mode, appearance count, and target output length. It evaluates 12 open-weight models across five families with primarily 1% poison budget, using attack success rate as the main metric.","area":"Safety & Trustworthiness","applicationDomains":["Cybersecurity","Transport & Logistics"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity","Logistics"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.23168","pdf":"https://arxiv.org/pdf/2605.23168","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.23168"},"evidence":{"snippet":"We introduce PoisonForge, a benchmark that parameterizes this threat along four dimensions (bias type, poisoning mode, appearance count, and target output length) and evaluates 12 open-weight models (from 2B to 32B parameters) across five families under a primarily 1% poison budget.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.23168"},"ranking":{},"description":"PoisonForge benchmarks task-level targeted poisoning of instruction-tuned LLMs, parameterizing bias type, poisoning mode, appearance count, and target output length. It evaluates 12 open-weight models across five families with primarily 1% poison budget, using attack success rate as the main metric.","whyItMatters":"Data supply chain poisoning poses a real threat when fine-tuning on unvetted data. PoisonForge quantifies vulnerability across models and configurations, highlighting that design choices rather than scale drive risk, aiding in risk assessment and mitigation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0b816d0d5409e4d3ebe05d45294a5e88bfd8eff3e42dcbb48db39ca9b9738f72"},"motivation":"When practitioners fine-tune LLMs on unvetted datasets, an adversary can exploit the data supply chain through task-level poisoning: inserting a small number of crafted instruction-response pairs that cause the model to embed attacker-specified entities, such as a country, in outputs for a targeted task family while behaving normally elsewhere.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.23168","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"cross-domain"},{"id":"bm_policyshiftbench_8ca9566e","familyId":"bmf_1a011cbd85dd","name":"PolicyShiftBench","oneLine":"PolicyShiftBench evaluates policy-adaptive image guardrailing: given an image and a current policy, a model must output a pass/block decision plus optional violated category IDs. The benchmark comprises 2,000 policy-discriminative instances over 265 images, each paired with multiple policy-conditioned prompts. Scoring uses binary pass/block accuracy and category attribution metrics.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.05910","pdf":"https://arxiv.org/pdf/2607.05910","project":null,"code":"https://github.com/ssmisya/PolicyShiftGuard","data":null,"hfPaper":"https://huggingface.co/papers/2607.05910"},"evidence":{"snippet":"We introduce PolicyShiftBench, a comprehensive benchmark with 2,000 policy-discriminative instances over 265 images, where each image is paired with 7.55 policy-conditioned prompts on average to test whether models adapt to the active policy rather than relying on image-level safety priors.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":38,"hfDailySubmittedAt":"2026-07-16T00:00:00.000Z","githubStars":22,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.05910"},"ranking":{"90d":{"score":48,"rank":80,"coverage":0.7,"confidence":"Medium"}},"description":"PolicyShiftBench evaluates policy-adaptive image guardrailing: given an image and a current policy, a model must output a pass/block decision plus optional violated category IDs. The benchmark comprises 2,000 policy-discriminative instances over 265 images, each paired with multiple policy-conditioned prompts. Scoring uses binary pass/block accuracy and category attribution metrics.","whyItMatters":"Existing image safety benchmarks assume safety is a fixed property of an image, whereas real deployments vary policies across products and regions. PolicyShiftBench measures whether models can bind image evidence to the active policy rather than relying on image-level priors, providing a practical evaluation for guardrails in dynamic policy environments.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"96af7e72f658f21967c3d891f085a7c6c3e49392ed4692678485a7b50b0f5cd2"},"motivation":"Image guardrails are typically trained and evaluated under a fixed safety policy, implicitly treating safety as an intrinsic property of an image.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.05910","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_polistemics_ea35d629","familyId":"bmf_14debe1298ea","name":"Polistemics","oneLine":"Polistemics evaluates LLMs as mediators of political information across controlled settings varying evidence clarity, noise, and consistency, using a diagnostic benchmark grounded in Epistemic Modesty.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25953","pdf":"https://arxiv.org/pdf/2607.25953","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.25953"},"evidence":{"snippet":"We introduce Polistemics, a theory-grounded diagnostic benchmark for evaluating LLMs as mediators of political information in elections.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25953"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Polistemics evaluates LLMs as mediators of political information across controlled settings varying evidence clarity, noise, and consistency, using a diagnostic benchmark grounded in Epistemic Modesty.","whyItMatters":"High aggregate scores can mask systematic failures in LLM political mediation, particularly under ambiguous or contradictory evidence, affecting citizens' ability to make informed electoral decisions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7aa2097eb092a875d1e16ec5079cbce1399c3380f441a114c6872db772b2e86b"},"motivation":"As LLMs increasingly shape the political information citizens rely on, no standard exists to assess whether they do so responsibly.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25953","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_polycomp_d2741a8f","familyId":"bmf_0fe1b33f5028","name":"PolyComp","oneLine":"PolyComp is a benchmark for compositional 3D spatial reasoning using polycube problems with 120 problems across four geometry families.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Geometric reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.14741","pdf":"https://arxiv.org/pdf/2608.14741","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.14741"},"evidence":{"snippet":"We introduce PolyComp, a procedurally generated and verified benchmark that stresses visual recognition and compositional spatial reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14741"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PolyComp is a benchmark for compositional 3D spatial reasoning using polycube problems with 120 problems across four geometry families.","whyItMatters":"It stresses visual recognition and compositional reasoning, with random guessing baseline at 25% for challenging multimodal evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4d0aea6734f85d47161b7b78f6452fb39a3d4dd39878bbd63e5491724c296b1f"},"motivation":"We introduce PolyComp, a procedurally generated and verified benchmark that stresses visual recognition and compositional spatial reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14741","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_51c52a821f9daa7a","familyId":"catalog_family_51c52a821f9daa7a","name":"PolyMath","oneLine":"Polymath is a challenging multi-modal mathematical reasoning benchmark designed to evaluate the general cognitive reasoning abilities of Multi-modal Large Language Models (MLLMs). The benchmark comprises 5,000 manually collected high-quality images of cognitive textual and visual challenges across 10 distinct categories, including pattern recognition, spatial reasoning, and relative reasoning.","description":"Polymath is a challenging multi-modal mathematical reasoning benchmark designed to evaluate the general cognitive reasoning abilities of Multi-modal Large Language Models (MLLMs). The benchmark comprises 5,000 manually collected high-quality images of cognitive textual and visual challenges across 10 distinct categories, including pattern recognition, spatial reasoning, and relative reasoning.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multilingual","Math","Multimodal","Reasoning","Spatial Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_51c52a821f9daa7a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/polymath"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/polymath"}],"catalogSources":[{"catalog":"benchlm","sourceId":"polyMath","url":"https://benchlm.ai/benchmarks/polymath","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"PolyMath","format":"Cross-lingual mathematical reasoning","tasks":"Multilingual math problems","successorKey":null},{"catalog":"llm-stats","sourceId":"polymath","url":"https://llm-stats.com/benchmarks/polymath","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multilingual","math","multimodal","reasoning","spatial reasoning","vision"],"catalogModelCount":23,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_a50f4a901acc5f37","familyId":"catalog_family_a50f4a901acc5f37","name":"PolyMath-en","oneLine":"PolyMath is a multilingual mathematical reasoning benchmark covering 18 languages and 4 difficulty levels from easy to hard, ensuring difficulty comprehensiveness, language diversity, and high-quality translation. The benchmark evaluates mathematical reasoning capabilities of large language models across diverse linguistic contexts, making it a highly discriminative multilingual mathematical benchmark.","description":"PolyMath is a multilingual mathematical reasoning benchmark covering 18 languages and 4 difficulty levels from easy to hard, ensuring difficulty comprehensiveness, language diversity, and high-quality translation. The benchmark evaluates mathematical reasoning capabilities of large language models across diverse linguistic contexts, making it a highly discriminative multilingual mathematical benchmark.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/polymath-en","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a50f4a901acc5f37"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/polymath-en"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"polymath-en","url":"https://llm-stats.com/benchmarks/polymath-en","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_polyspeech-100_ce0b7268","familyId":"bmf_0150fcfdb2d3","name":"PolySpeech-100","oneLine":"PolySpeech-100 evaluates speech understanding in speech-large language models across 110 linguistic variants, including 19 Chinese dialects and over 80 low-resource languages, using tasks that assess semantic reasoning beyond transcription.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01016","pdf":"https://arxiv.org/pdf/2606.01016","project":null,"code":"https://github.com/YoungSeng/PolySpeech-100","data":null,"hfPaper":"https://huggingface.co/papers/2606.01016"},"evidence":{"snippet":"To bridge this gap, we introduce PolySpeech-100, a massive-scale benchmark designed to assess `native-level' speech comprehension across 110 linguistic variants.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01016"},"ranking":{},"description":"PolySpeech-100 evaluates speech understanding in speech-large language models across 110 linguistic variants, including 19 Chinese dialects and over 80 low-resource languages, using tasks that assess semantic reasoning beyond transcription.","whyItMatters":"Existing speech benchmarks are biased toward high-resource languages and focus on low-level recognition, limiting assessment of reasoning abilities and dialect robustness. PolySpeech-100 provides a broader coverage and a scoring protocol for comparing model performance on diverse speech understanding tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e48faa79f37ee5f2f409c8154817d35081d1eb784b978a44cea4621c957ad93e"},"motivation":"While End-to-End (E2E) Speech-Large Language Models (Speech-LLMs) are rapidly evolving, their evaluation methodologies remain limited to the era of simple transcription.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01016","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"PolySpeech-100 Benchmark Team","organizationType":"academic-lab","sourceUrl":"https://github.com/YoungSeng/PolySpeech-100","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_polyworkbench_2abf54e5","familyId":"bmf_91789172102b","name":"PolyWorkBench","oneLine":"PolyWorkBench evaluates LLM agents on multilingual, long-horizon workplace workflows across five domains: commerce, knowledge work, legal analysis, localization, and manufacturing. It includes 67 tasks, structured scoring via Grade, executable state verification with Pytest, and LLM-as-Judge diagnostics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.06008","pdf":"https://arxiv.org/pdf/2607.06008","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.06008"},"evidence":{"snippet":"We introduce PolyWorkBench, a benchmark designed to evaluate LLM agents on multilingual, long-horizon workplace workflows.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06008"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"PolyWorkBench evaluates LLM agents on multilingual, long-horizon workplace workflows across five domains: commerce, knowledge work, legal analysis, localization, and manufacturing. It includes 67 tasks, structured scoring via Grade, executable state verification with Pytest, and LLM-as-Judge diagnostics.","whyItMatters":"This benchmark fills the gap of evaluating LLM agents on tasks that combine multilinguality and long-horizon execution, providing a structured protocol to compare agent performance and identify systematic failure modes in cross-lingual scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"40e85edc51a0543817b5765dc2b8757fba2da41c8f8d09fa6b86e7a1e309dcee"},"motivation":"While Large Language Model (LLM) agents excel at monolingual long-horizon planning and tool use, enterprise workflows inherently require processing multilingual resources across extended trajectories.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06008","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"PolyWorkBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.06008","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_poolbench_e79b8c50","familyId":"bmf_9706d0afa65d","name":"PoolBench","oneLine":"PoolBench isolates pooling strategies as the experimental variable in concept representation evaluation for decoder-only LLMs. It covers 17 concepts, 19 pooling strategies, and 3 open-weight models (Llama-3.1-8B, Gemma-2-9B, Mistral-7B) on a corpus of 37,693 real-text passages, with primary axis linear separability (AUROC) and diagnostic axes for steering and disentanglement.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.05162","pdf":"https://arxiv.org/pdf/2608.05162","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.05162"},"evidence":{"snippet":"We introduce PoolBench, a benchmark that isolates pooling as the experimental variable under a fixed evaluation protocol.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.05162"},"ranking":{},"description":"PoolBench isolates pooling strategies as the experimental variable in concept representation evaluation for decoder-only LLMs. It covers 17 concepts, 19 pooling strategies, and 3 open-weight models (Llama-3.1-8B, Gemma-2-9B, Mistral-7B) on a corpus of 37,693 real-text passages, with primary axis linear separability (AUROC) and diagnostic axes for steering and disentanglement.","whyItMatters":"Pooling is a consequential but under-examined design choice in concept representation work, yet no shared protocol exists for comparing pooling rules across concepts, models, and tasks. PoolBench provides a controlled protocol with released corpus, pre-extracted activations, and scoring code to enable principled comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b1f2e70efdb1ba23147bcbbf582544f71de38918836a8857adc217b0714068bb"},"motivation":"Pooling is a consequential but under-examined design choice in decoder-only concept representation work: practitioners must collapse token-level hidden states into a passage-level vector, yet no shared protocol exists for comparing this choice across concepts, models, and tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.05162","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_003394ba687ec677","familyId":"catalog_family_003394ba687ec677","name":"POPE","oneLine":"Polling-based Object Probing Evaluation (POPE) is a benchmark for evaluating object hallucination in Large Vision-Language Models (LVLMs). POPE addresses the problem where LVLMs generate objects inconsistent with target images by using a polling-based query method that asks yes/no questions about object presence in images, providing more stable and flexible evaluation of object hallucination.","description":"Polling-based Object Probing Evaluation (POPE) is a benchmark for evaluating object hallucination in Large Vision-Language Models (LVLMs). POPE addresses the problem where LVLMs generate objects inconsistent with target images by using a polling-based query method that asks yes/no questions about object presence in images, providing more stable and flexible evaluation of object hallucination.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Safety","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/pope","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_003394ba687ec677"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/pope"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"pope","url":"https://llm-stats.com/benchmarks/pope","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","safety","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_95324dafb4b28b23","familyId":"catalog_family_95324dafb4b28b23","name":"PopQA","oneLine":"PopQA is an entity-centric open-domain question-answering dataset consisting of 14,000 QA pairs designed to evaluate language models' ability to memorize and recall factual knowledge across entities with varying popularity levels. The dataset probes both parametric memory (stored in model parameters) and non-parametric memory effectiveness, with questions covering 16 diverse relationship types from Wikidata converted to natural language using templates. Created by sampling knowledge triples from Wikidata and converting them to natural language questions, focusing on long-tail entities to understand LMs' strengths and limitations in memorizing factual knowledge.","description":"PopQA is an entity-centric open-domain question-answering dataset consisting of 14,000 QA pairs designed to evaluate language models' ability to memorize and recall factual knowledge across entities with varying popularity levels. The dataset probes both parametric memory (stored in model parameters) and non-parametric memory effectiveness, with questions covering 16 diverse relationship types from Wikidata converted to natural language using templates. Created by sampling knowledge triples from Wikidata and converting them to natural language questions, focusing on long-tail entities to understand LMs' strengths and limitations in memorizing factual knowledge.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/popqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_95324dafb4b28b23"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/popqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"popqa","url":"https://llm-stats.com/benchmarks/popqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_popsicle_93cd1103","familyId":"bmf_39a63441d299","name":"POPSICLE","oneLine":"POPSICLE benchmarks cryoET segmentation and macromolecular localization using data from the CryoET Data Portal. It spans eukaryotic and prokaryotic systems, purified and in situ samples, and covers dense voxel-wise segmentation and sparse localization tasks. Built on the CryoET Data Portal, it can expand with new data.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["eess.IV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.10255","pdf":"https://arxiv.org/pdf/2606.10255","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.10255"},"evidence":{"snippet":"Here, we present POPSICLE, a benchmark suite for cryoET segmentation and macromolecular localization built from the CryoET Data Portal - an open, ML-ready repository of tomographic data, metadata, and annotations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.10255"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"POPSICLE benchmarks cryoET segmentation and macromolecular localization using data from the CryoET Data Portal. It spans eukaryotic and prokaryotic systems, purified and in situ samples, and covers dense voxel-wise segmentation and sparse localization tasks. Built on the CryoET Data Portal, it can expand with new data.","whyItMatters":"CryoET lacks standardized, well-annotated benchmarks, limiting robust comparison across methods. POPSICLE provides an open, extensible foundation from a living repository, with baseline experiments showing task-dependent model rankings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"507b64bb6c98fe6e010d730383e7ab831f7b5231c7681db69459252ece1bf400"},"motivation":"Cryo-electron tomography (cryoET) has emerged as a powerful tool in structural and cellular biology by enabling direct visualization of macromolecular structures within intact cells, thereby linking molecular architecture to cellular organization in a native context.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.10255","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_portbench_32129f61","familyId":"bmf_1aa870031c3f","name":"PortBench","oneLine":"PortBench evaluates LLM-driven portfolio management via a static QA dataset (6,269 questions across seven task templates) and a dynamic five-stage allocation pipeline, spanning six asset classes over ten years. Scoring includes a dual-layer correlation score and CEPS, with evaluation under three stress regimes and investor profiles.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27887","pdf":"https://arxiv.org/pdf/2605.27887","project":"https://portbench.github.io/","code":"https://github.com/AgenticFinLab/portbench","data":null,"hfPaper":"https://huggingface.co/papers/2605.27887"},"evidence":{"snippet":"We introduce PortBench, a benchmark spanning six heterogeneous asset classes over ten years.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":8,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27887"},"ranking":{},"description":"PortBench evaluates LLM-driven portfolio management via a static QA dataset (6,269 questions across seven task templates) and a dynamic five-stage allocation pipeline, spanning six asset classes over ten years. Scoring includes a dual-layer correlation score and CEPS, with evaluation under three stress regimes and investor profiles.","whyItMatters":"Existing financial benchmarks often ignore cross-asset correlations and the full portfolio management pipeline. PortBench addresses this gap by providing a comprehensive, reusable evaluation for LLM capabilities in realistic portfolio management, enabling comparison across models and informing practical deployment decisions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b514623ab90b9974c54f11299d7b16c95e80565ce3ccbe1091959579bfd21376"},"motivation":"Large language models (LLMs) have shown strong performance across diverse financial tasks, yet portfolio management (PM) remains poorly benchmarked.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27887","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AgenticFinLab","organizationType":"academic-lab","sourceUrl":"https://github.com/AgenticFinLab/portbench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_portexto_2594bd31","familyId":"bmf_89b15b527989","name":"PorTEXTO","oneLine":"PorTEXTO evaluates visual text extraction from contemporary and culturally relevant European Portuguese (pt-PT) images. The benchmark includes synthetic and real-world samples with native-speaker-reviewed transcriptions.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19096","pdf":"https://arxiv.org/pdf/2606.19096","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.19096"},"evidence":{"snippet":"This work addresses modern OCR applications, introducing PorTEXTO, the first benchmark for contemporary and culturally relevant pt-PT visual text extraction.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19096"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"PorTEXTO evaluates visual text extraction from contemporary and culturally relevant European Portuguese (pt-PT) images. The benchmark includes synthetic and real-world samples with native-speaker-reviewed transcriptions.","whyItMatters":"European Portuguese is underrepresented in OCR benchmarks, which typically focus on high-resource languages or historical documents. PorTEXTO addresses this gap by providing a modern pt-PT evaluation set, enabling assessment of OCR models for contemporary applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0e8448455321582b00d1e3fef0df910c77cef5dd7a68988025a3100498bbf48c"},"motivation":"European Portuguese (pt-PT) is largely absent from OCR benchmarks, which skew toward high-resource languages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19096","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_benchmarking-automated-security-patch-back_bac96e83","familyId":"bmf_47f9bd1f1452","name":"Porting Benchmark","oneLine":"Porting Benchmark evaluates automated security patch backporting using 1,234 curated cases across cross-version, cross-branch, and cross-repository scenarios, with a common evaluation framework and dynamic validation on a subset.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.17671","pdf":"https://arxiv.org/pdf/2608.17671","project":"https://doi.org/10.5281/zenodo.21785770","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present Porting Benchmark, a curated dataset of 1,234 security patch backporting cases spanning cross-version, cross-branch, and cross-repository scenarios, paired with a common evaluation framework.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17671"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Porting Benchmark evaluates automated security patch backporting using 1,234 curated cases across cross-version, cross-branch, and cross-repository scenarios, with a common evaluation framework and dynamic validation on a subset.","whyItMatters":"The benchmark aligns evaluation conditions across five tools and reveals substantial performance degradation on complex patches, highlighting the need for generalizable backporting approaches and realistic executable validation.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"05486539de8349aee234d34e43920a768f5db57853f5331184803cc3146ab042"},"motivation":"Automated security patch backporting is critical for mitigating N-day vulnerabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is explicitly named, has a curated dataset and common evaluation framework, and the Artifact DOI is provided as a public path.","canonicalNameSource":"abstract","canonicalNameEvidence":"We present Porting Benchmark, a curated dataset of 1,234 security patch backporting cases spanning cross-version, cross-branch, and cross-repository scenarios, paired with a common evaluation framework."},"publication":{"status":"acceptance_claimed","venue":"ASE 2026","evidence":"13 pages, 3 figures. Accepted at ASE 2026. Artifact: https://doi.org/10.5281/zenodo.21785770","evidenceUrl":"https://arxiv.org/abs/2608.17671","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-19T11:05:13.395391Z"},"venueAttempts":[{"venueName":"ASE 2026","reviewStatus":"accepted","decisionRaw":"13 pages, 3 figures. Accepted at ASE 2026. Artifact: https://doi.org/10.5281/zenodo.21785770","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.17671","observedAt":"2026-08-19T11:05:13.395391Z","rawValue":"13 pages, 3 figures. Accepted at ASE 2026. Artifact: https://doi.org/10.5281/zenodo.21785770","level":"author-claim"}]}],"attentionForecast":{"score":58,"confidence":"Medium","horizon":"7d","reason":"The benchmark is accepted at ASE 2026 with a public artifact and addresses a practical security automation problem, likely drawing interest from software engineering and security communities."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_posterbench_1bd5f76e","familyId":"bmf_0e909d54ed40","name":"PosterBench","oneLine":"PosterBench evaluates academic paper-to-poster generation. It includes a 100-paper Main Track across five disciplines and a 10-paper mini subset, with automated scoring and human evaluation.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.13560","pdf":"https://arxiv.org/pdf/2608.13560","project":null,"code":"https://github.com/Yaxin9Luo/AutoDesign","data":null,"hfPaper":null},"evidence":{"snippet":"To instantiate and evaluate this framework, we focus on the academic paper-to-poster generation task and introduce PosterBench, comprising a 100-paper Main Track spanning five disciplines and PosterBench-mini, a shared 10-paper subset for controlled evaluation.","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence","public artifact URL","named entity is a model, method, framework, or dataset used in evaluation"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":55,"hfDailySubmittedAt":"2026-08-14T00:00:00.000Z","githubStars":182,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13560"},"ranking":{"30d":{"score":72,"rank":4,"coverage":0.85,"confidence":"High"},"90d":{"score":65,"rank":12,"coverage":0.7,"confidence":"Medium"}},"description":"PosterBench evaluates academic paper-to-poster generation. It includes a 100-paper Main Track across five disciplines and a 10-paper mini subset, with automated scoring and human evaluation.","whyItMatters":"It provides a standardized way to compare agentic design systems on long-horizon multimodal generation, enabling measurement of quality and agent efficiency across different models and harnesses.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"79f26d0006e8da6f0f68ec259e918a9d462e1bb44b089caeee2cd7b6ba29a583"},"motivation":"Transforming multimodal sources into condensed and structured media outputs can be fundamentally conceptualized as a long-horizon agentic process centered on a model-harness system.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"PosterBench is introduced as a benchmark with a defined evaluation protocol and public dataset, enabling comparable scoring across systems.","canonicalNameSource":"abstract","canonicalNameEvidence":"we focus on the academic paper-to-poster generation task and introduce PosterBench, comprising a 100-paper Main Track spanning five disciplines and PosterBench-mini, a shared 10-paper subset for controlled evaluation."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13560","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"The benchmark is tied to a novel framework with public code and dataset links, and the task of paper-to-poster generation has moderate community interest."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_7a593ca454f1246a","familyId":"catalog_family_7a593ca454f1246a","name":"PostTrain Bench","oneLine":"PostTrainBench evaluates a model's ability to autonomously post-train base models. Given pretrain-only base models, the agent must complete the full pipeline of data synthesis, training, evaluation, and iteration within a time budget, scored across downstream benchmarks such as AIME2025, BFCL, GPQA Main, GSM8K, and HumanEval.","description":"PostTrainBench evaluates a model's ability to autonomously post-train base models. Given pretrain-only base models, the agent must complete the full pipeline of data synthesis, training, evaluation, and iteration within a time budget, scored across downstream benchmarks such as AIME2025, BFCL, GPQA Main, GSM8K, and HumanEval.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Reasoning","Agents","Code","Systems"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7a593ca454f1246a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/posttrainbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/posttrainbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"postTrainBench","url":"https://benchlm.ai/benchmarks/posttrainbench","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"PostTrain Bench","format":"Harbor agent evaluation","tasks":"Post-training software-engineering tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"posttrainbench","url":"https://llm-stats.com/benchmarks/posttrainbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","reasoning","agents","code","systems"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_93924c7b9f3ba935","familyId":"catalog_family_93924c7b9f3ba935","name":"PostTrainBench Lite","oneLine":"PostTrainBench Lite measures whether an agent can design and execute a full post-training strategy (data, prompts, RL recipe, and eval loop) for a pretrained base model under a constrained time budget, scored as normalized mean reward over the improvement window.","description":"PostTrainBench Lite measures whether an agent can design and execute a full post-training strategy (data, prompts, RL recipe, and eval loop) for a pretrained base model under a constrained time budget, scored as normalized mean reward over the improvement window.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Agents","Code","Systems"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/posttrainbench-lite","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_93924c7b9f3ba935"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/posttrainbench-lite"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"posttrainbench-lite","url":"https://llm-stats.com/benchmarks/posttrainbench-lite","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","code","systems"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_power-systems-agent-benchmark_68561f3f","familyId":"bmf_9170ad13958e","name":"Power Systems Agent Benchmark","oneLine":"Power Systems Agent Benchmark is an executable benchmark for power-engineering agents. It includes 41 task families with deterministic evaluators that recompute engineering quantities and check constraints, plus held-out generation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.20950","pdf":"https://arxiv.org/pdf/2606.20950","project":"https://doi.org/10.5281/zenodo.20753046","code":"https://github.com/trashchenkov/power-systems-agent-benchmark","data":null,"hfPaper":"https://huggingface.co/papers/2606.20950"},"evidence":{"snippet":"We introduce the Power Systems Agent Benchmark, an executable benchmark for power-engineering agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.20950"},"ranking":{"90d":{"score":25,"rank":299,"coverage":0.7,"confidence":"Medium"}},"description":"Power Systems Agent Benchmark is an executable benchmark for power-engineering agents. It includes 41 task families with deterministic evaluators that recompute engineering quantities and check constraints, plus held-out generation.","whyItMatters":"Executable evaluation in power engineering is missing. This benchmark provides a repeatable, contamination-resistant protocol for tool-using agents, with public and hidden splits.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b76a1ac6a6646826cbebd5ce468edd53af9f5493b023d78661b8b1792ffa1be2"},"motivation":"Executable evaluation -- checking the consequences of an agent's actions with a program rather than grading its prose -- has become a prominent way to assess tool-using AI agents in software settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.20950","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Power Systems Agent Benchmark Team","organizationType":"academic-lab","sourceUrl":"https://github.com/trashchenkov/power-systems-agent-benchmark","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_powercodebench_c0ce0f67","familyId":"bmf_dc78daa563c7","name":"PowerCodeBench","oneLine":"PowerCodeBench is a benchmark generator for power system code generation, paired with an intervention method, but no artifacts are provided in this article.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation","Factuality"],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.31478","pdf":"https://arxiv.org/pdf/2605.31478","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.31478"},"evidence":{"snippet":"We introduce PowerCodeBench, an execution-validated benchmark generator that pairs natural-language operator queries with pandapower code and numerical ground truth; an L0-L3 documentation-driven probing procedure that measures per-model API knowledge profiles; and a boundary-aware intervention that combines query-side API demand estimation with targeted proactive documentation injection and routed reactive correction.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.31478"},"ranking":{},"description":"PowerCodeBench is a benchmark generator for power system code generation, paired with an intervention method, but no artifacts are provided in this article.","whyItMatters":"It addresses reliability of open-weight models for on-premise deployment, but the lack of a release prevents external use.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"db350dac8e1dfdd93cb56df1e7b068a6448279e388e3944139dc9fe5222260fc"},"motivation":"Large language models (LLMs) are increasingly used to automate power-system analysis, but many utilities and energy-research labs require on-premise serving for confidentiality, regulatory, reproducibility, and cost reasons.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.31478","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_ppe-bench_ab4b6b83","familyId":"bmf_74f307632439","name":"PPE-Bench","oneLine":"PPE-Bench evaluates machine unlearning in multimodal large language models under private-public entanglement, where images contain a target individual to forget and public elements to preserve.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.02897","pdf":"https://arxiv.org/pdf/2607.02897","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.02897"},"evidence":{"snippet":"To address these limitations, we propose PPE-Bench, a new benchmark for evaluating MLLM unlearning under private-public entanglement.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02897"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PPE-Bench evaluates machine unlearning in multimodal large language models under private-public entanglement, where images contain a target individual to forget and public elements to preserve.","whyItMatters":"Addresses the lack of benchmarks that reflect real-world image complexity and entanglement of private and public information, supporting evaluation of unlearning methods that must preserve public context.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0b6b09e58576b7f26989245ac1e30cc6b20463c1104f41034d141ce4992e1797"},"motivation":"Multimodal Large Language Models (MLLMs) have shown strong capabilities, but they may memorize private information from web data, raising privacy concerns.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"EMNLP 2026","evidence":"to appear in EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2607.02897","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"EMNLP 2026","reviewStatus":"accepted","decisionRaw":"to appear in EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.02897","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"to appear in EMNLP 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ppt-eval_9317e9e6","familyId":"bmf_32aec5f2ab84","name":"PPT-Eval","oneLine":"PPT-Eval is a benchmark of 120 PowerPoint tasks across 12 files for computer-use agents. It covers content creation and editing, with rubric-based evaluation that awards partial credit and provides natural language feedback.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.31154","pdf":"https://arxiv.org/pdf/2606.31154","project":"https://microsoft.github.io/ppteval","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.31154"},"evidence":{"snippet":"We introduce PPT-Eval, a benchmark of 120 PowerPoint tasks across 12 files that cover both content creation and presentation editing scenarios, organized by difficulty.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31154"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PPT-Eval is a benchmark of 120 PowerPoint tasks across 12 files for computer-use agents. It covers content creation and editing, with rubric-based evaluation that awards partial credit and provides natural language feedback.","whyItMatters":"Provides a realistic, multimodal testbed for computer-use agents. The rubric-based scoring captures partial progress and correlates with human judgment, enabling nuanced comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5a27f5023da02e575c3b0c42ea5523c37770d4f14da9bd97bdf8f660b8fd44ab"},"motivation":"Creating and editing slides is a rich, multimodal activity that is ubiquitous in professional and educational settings, making it an ideal testbed for real-world computer-use agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31154","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Microsoft","organizationType":"company-research-lab","sourceUrl":"https://microsoft.github.io/ppteval","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_pragmatch_5cfd2fc5","familyId":"bmf_c5b11c599fa8","name":"PragMatch","oneLine":"PragMatch is a controlled set of 3,000 image-text pairs derived from MMSD2.0 for studying pragmatic incongruity in multimodal sarcasm detection, with original sarcastic examples and constructed literal and hard-negative pairs.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09772","pdf":"https://arxiv.org/pdf/2608.09772","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09772"},"evidence":{"snippet":"We introduce PragMatch, a controlled benchmark of 3,000 image-text pairs derived from MMSD2.0, including original sarcastic examples and constructed literal and hard-negative pairs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09772"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PragMatch is a controlled set of 3,000 image-text pairs derived from MMSD2.0 for studying pragmatic incongruity in multimodal sarcasm detection, with original sarcastic examples and constructed literal and hard-negative pairs.","whyItMatters":"This resource helps investigate whether large vision-language models rely on superficial cues instead of genuine reasoning in multimodal sarcasm, highlighting practical limitations in model evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"02b3858c0d5e91311baa7320036df25cb670bac676f468442886c33674339fa7"},"motivation":"Large Vision-Language Models (LVLMs) have demonstrated strong performance on multimodal benchmarks, yet it remains unclear whether they genuinely reason about relationships between images and text or rely on superficial correlations, known as shortcut learning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09772","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_44c9e16983fcb045","familyId":"catalog_family_44c9e16983fcb045","name":"PRBench-Finance","oneLine":"PRBench-Finance evaluates professional reasoning on finance tasks.","description":"PRBench-Finance evaluates professional reasoning on finance tasks.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","Finance"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/prbench-finance","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_44c9e16983fcb045"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/prbench-finance"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"prbench-finance","url":"https://llm-stats.com/benchmarks/prbench-finance","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","finance"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_780cd471a7228fcb","familyId":"catalog_family_780cd471a7228fcb","name":"PRBench-Legal","oneLine":"PRBench-Legal evaluates professional reasoning on legal tasks.","description":"PRBench-Legal evaluates professional reasoning on legal tasks.","area":"Language & Knowledge","applicationDomains":["Law & Government"],"primaryDomain":"Law & Government","industrySectors":[],"capabilities":[],"topics":["Knowledge","Legal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/prbench-legal","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_780cd471a7228fcb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/prbench-legal"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"prbench-legal","url":"https://llm-stats.com/benchmarks/prbench-legal","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","legal","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_pre-flight_898fa920","familyId":"bmf_6f851dc66002","name":"Pre-Flight","oneLine":"Pre-Flight evaluates large language models on aviation operational knowledge via 300 multiple-choice questions drawn from international standards and airport ground operations material, covering ground operations, ICAO and FAA regulations, general aviation knowledge, and operational scenarios. Scoring is by accuracy under a standard multiple-choice protocol using the Inspect framework.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.01829","pdf":"https://arxiv.org/pdf/2607.01829","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.01829"},"evidence":{"snippet":"We present Pre-Flight, an open source benchmark of 300 multiple choice questions drawn from international standards and airport ground operations material, covering international airport ground operations, ICAO and US FAA regulations, aviation general knowledge and complex operational scenarios.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01829"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Pre-Flight evaluates large language models on aviation operational knowledge via 300 multiple-choice questions drawn from international standards and airport ground operations material, covering ground operations, ICAO and FAA regulations, general aviation knowledge, and operational scenarios. Scoring is by accuracy under a standard multiple-choice protocol using the Inspect framework.","whyItMatters":"General-purpose benchmarks do not assess aviation-specific operational safety knowledge, a high-stakes domain where incorrect reasoning can have serious consequences. This benchmark provides a domain-specific evaluation to gauge model reliability for non-safety-critical aviation operations.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"332b08eb0dc49f79dece7a81c9299b4606050beeee64e140a8bdfc4a9c611587"},"motivation":"Large language models (LLMs) are increasingly proposed for aviation business operations, from documentation and training generation to customer facing assistants.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01829","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_preact-bench_d5c8c56f","familyId":"bmf_cb7720730271","name":"PreAct-Bench","oneLine":"PreActBench is a benchmark for predictive monitoring in LLMs, consisting of 1,000 paired ethical and unethical action trajectories across five domains. It evaluates whether models can infer if a partial trajectory will culminate in unethical action, using the Prefix Foresight F1 metric.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09890","pdf":"https://arxiv.org/pdf/2606.09890","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.09890"},"evidence":{"snippet":"To support this task, we present PreActBench, a benchmark of 1,000 paired ethical and unethical action trajectories spanning five domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09890"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PreActBench is a benchmark for predictive monitoring in LLMs, consisting of 1,000 paired ethical and unethical action trajectories across five domains. It evaluates whether models can infer if a partial trajectory will culminate in unethical action, using the Prefix Foresight F1 metric.","whyItMatters":"Safety research often detects unethical behavior only after it occurs. Predictive monitoring enables anticipation of harm before execution, and PreActBench measures this capability across models and guardrails.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"476be280857a1820deeefed7ec9eea374221783707fbbf60ebdd8f2328a60cb3"},"motivation":"Large language models (LLMs) are increasingly deployed as autonomous agents capable of executing multi-step action trajectories toward a given objective.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09890","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_predact-bench_f35f1a5f","familyId":"bmf_9ef627ddae02","name":"PredAct-Bench","oneLine":"PredAct-Bench evaluates dialogue agents paired with imperfect tools using educational datasets, measuring AI-assisted decision-making and trust metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02372","pdf":"https://arxiv.org/pdf/2608.02372","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.02372"},"evidence":{"snippet":"We introduce PREDACTBENCH, a benchmark for evaluating dialogue agents paired with statistically imperfect tools, using education as a measurable testbed where ground truth outcomes and clear intervention decisions are available.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02372"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PredAct-Bench evaluates dialogue agents paired with imperfect tools using educational datasets, measuring AI-assisted decision-making and trust metrics.","whyItMatters":"Highlights gap in existing benchmarks that assume perfect tool reliability, important for high-stakes domains.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5aa412499027fa3711c2385a3edf655b6ddc1e651674a4a81be79947ca4f7618"},"motivation":"Large Language Models (LLMs) are increasingly deployed in task-oriented dialogue systems that support multi-step decision-making in high-stakes domains such as education, healthcare, and finance.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02372","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_def7139ed7186deb","familyId":"catalog_family_def7139ed7186deb","name":"PresentBench","oneLine":"PresentBench evaluates AI agents on producing presentation-style deliverables, such as generating lesson-plan slides and structured documents from source materials.","description":"PresentBench evaluates AI agents on producing presentation-style deliverables, such as generating lesson-plan slides and structured documents from source materials.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/presentbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_def7139ed7186deb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/presentbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"presentbench","url":"https://llm-stats.com/benchmarks/presentbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_principle-bench_119ba0f5","familyId":"bmf_dd484486344c","name":"Principle-Bench","oneLine":"Principle-Bench contains 168 cryptoasset financial-promotion scenarios mapped to two UK FCA principles, with paraphrase, adversarial keyword-stuffing, and boundary perturbations for evaluating LLM-as-judge on accuracy, paraphrase robustness, adversarial robustness, and calibration.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.CR"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.14329","pdf":"https://arxiv.org/pdf/2608.14329","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.14329"},"evidence":{"snippet":"We release Principle-Bench, 168 cryptoasset financial-promotion scenarios mapped to two UK FCA principles, with paraphrase, adversarial keyword-stuffing, and boundary perturbations authored under a pre-registered rubric; the first benchmark covering all four axes for principle-based regulation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14329"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Principle-Bench contains 168 cryptoasset financial-promotion scenarios mapped to two UK FCA principles, with paraphrase, adversarial keyword-stuffing, and boundary perturbations for evaluating LLM-as-judge on accuracy, paraphrase robustness, adversarial robustness, and calibration.","whyItMatters":"Addresses the evaluation gap for LLM-as-judge in principle-based regulation, where standards are not binary. Provides a multi-axis assessment to inform deployment decisions, as no single method dominates all axes and adversarial inputs can significantly degrade performance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7a8be2a0d792a4609acdb74afa5db195fc0626601ae2ac368b3a535b30201848"},"motivation":"Principle-based regulation, with evaluative standards such as \"fair, clear, and not misleading\" or \"deliver good outcomes\", cannot be reduced to binary predicates, and LLM-as-judge is increasingly used as the substitute.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"KDD 2026 Workshop on Secure and Trustworthy Large Language Models (SeT-LLM), poster","evidence":"7 pages, 3 figures. Accepted at the KDD 2026 Workshop on Secure and Trustworthy Large Language Models (SeT-LLM), poster","evidenceUrl":"https://arxiv.org/abs/2608.14329","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"KDD 2026 Workshop on Secure and Trustworthy Large Language Models (SeT-LLM), poster","reviewStatus":"accepted","decisionRaw":"7 pages, 3 figures. Accepted at the KDD 2026 Workshop on Secure and Trustworthy Large Language Models (SeT-LLM), poster","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.14329","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"7 pages, 3 figures. Accepted at the KDD 2026 Workshop on Secure and Trustworthy Large Language Models (SeT-LLM), poster","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_prionner_f3eb1b08","familyId":"bmf_80133507f6df","name":"PrionNER","oneLine":"Named entity recognition dataset from PubMed abstracts on prion disease, with 317 abstracts annotated for 15 coarse and 31 fine-grained entity types, plus evaluation scripts and train/test splits.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2605.28375","pdf":"https://arxiv.org/pdf/2605.28375","project":null,"code":"https://github.com/daotuanan/PrionNER/","data":null,"hfPaper":"https://huggingface.co/papers/2605.28375"},"evidence":{"snippet":"We benchmark supervised BERT baselines, W2NER, and zero-shot extractors on PrionNER.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28375"},"ranking":{},"description":"Named entity recognition dataset from PubMed abstracts on prion disease, with 317 abstracts annotated for 15 coarse and 31 fine-grained entity types, plus evaluation scripts and train/test splits.","whyItMatters":"Fills a gap in biomedical NLP for rare diseases, enabling evaluation of fine-grained and discontinuous entity extraction under low-resource conditions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f9ed7bfe78d23c31f858c49476b9b514b7d6b385da238a5c14cf2dc0005a9c9a"},"motivation":"Prion diseases are rare, rapidly progressive, and fatal neurodegenerative disorders that remain difficult to diagnose, particularly in their early stages because of nonspecific clinical presentations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ACL 25th Workshop on Biomedical Language Processing (BioNLP 2026)","evidence":"29 pages, 5 figures, accepted at ACL 25th Workshop on Biomedical Language Processing (BioNLP 2026)","evidenceUrl":"https://arxiv.org/abs/2605.28375","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ACL 25th Workshop on Biomedical Language Processing (BioNLP 2026)","reviewStatus":"accepted","decisionRaw":"29 pages, 5 figures, accepted at ACL 25th Workshop on Biomedical Language Processing (BioNLP 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2605.28375","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"29 pages, 5 figures, accepted at ACL 25th Workshop on Biomedical Language Processing (BioNLP 2026)","level":"author-claim"}]}],"publishers":[{"name":"PrionNER team","organizationType":"community","sourceUrl":"https://github.com/daotuanan/PrionNER/","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_prism_da9ef810","familyId":"bmf_bb4c8eb92b6e","name":"PRISM","oneLine":"Provides multimodal robot demonstrations and real-world evaluation tasks for precision, contact-rich industrial manipulation.","area":"Robotics & Embodied AI","applicationDomains":["Industrial & Engineering","Robotics & Autonomous Systems"],"primaryDomain":"Industrial & Engineering","industrySectors":["Manufacturing","Robotics"],"capabilities":["Multimodal robot manipulation","Contact-rich control","Imitation learning","Force-aware manipulation"],"topics":["Robotics","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.17962","pdf":"https://arxiv.org/pdf/2608.17962","project":"https://tengbo-yu.github.io/PRISM/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.17962"},"evidence":{"snippet":"In contrast to datasets collected in household or laboratory settings, PRISM provides a realistic benchmark for multimodal perception and control under high-precision industrial constraints, and serves as a foundation for contact-rich, generalizable manipulation in real-world manufacturing environments.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17962"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PRISM is a dataset of over 5,000 trajectories across 25 industrial manipulation tasks with multimodal sensing. It provides teleoperated demonstrations for contact-rich operations but lacks a standardized evaluation protocol or scoring contract.","whyItMatters":"As a dataset, it could support research in contact-rich manipulation, but without a scoring mechanism it does not constitute a benchmark for model comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f00b035dd00ad97000c398ba609017d1a1aebd893b7811b59f8ce360c77253c8"},"motivation":"Recent progress in robotic learning has been fueled by large-scale datasets collected in everyday environments.","constructionDetail":"PRISM covers precision industrial manipulation using synchronized RGB-D, force/torque, tactile and robot-state observations.","detail":{"taskBreakdown":["Precision contact-rich assembly","Product packaging","Dynamic object sorting","Force-aware manipulation"],"protocol":{"tasks":"25+ tasks and 5,000+ trajectories","primaryMetric":"Task success rate over 20 real-robot trials per configuration","version":"arXiv v1"}},"curation":{"state":"source-reviewed","reviewedAt":"2026-08-20","sources":["https://arxiv.org/abs/2608.17962","https://arxiv.org/html/2608.17962","https://tengbo-yu.github.io/PRISM/"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17962","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"score_submission","availability":{"paperStatus":"available","githubStatus":"announced_not_released","hfDatasetStatus":"announced_not_released","evaluatorStatus":"described_not_released","submissionStatus":"not_found"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"cross-domain"},{"id":"bm_prive-bench_ef1c909e","familyId":"bmf_cb1395d6b84b","name":"PriVE-Bench","oneLine":"PriVE-Bench evaluates vision-language models' visual grounding using paired original and counterfactual images, with PriVE-Tools extending to tool-derived evidence.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-14","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.16311","pdf":"https://arxiv.org/pdf/2607.16311","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.16311"},"evidence":{"snippet":"We introduce PriVE-Bench, a Prior-vs-Visual Evidence Benchmark that uses paired original and counterfactual images to distinguish visually grounded answers from prior-consistent errors.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.16311"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PriVE-Bench evaluates vision-language models' visual grounding using paired original and counterfactual images, with PriVE-Tools extending to tool-derived evidence.","whyItMatters":"Assesses whether vision-language models rely on learned priors rather than image content, and whether additional visual tools can improve grounding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fc31832d396eb7c618b0e2fa575e4b089072b1d2191f6326f06fd96b735efeba"},"motivation":"Vision-language models (VLMs) often answer visual questions using learned language and category priors rather than grounding their predictions in the image itself.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.16311","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_prm-as-a-judge-1-5-a-toolkit-for-robot-pro_42c65f99","familyId":"bmf_3a1382624b18","name":"PRM-as-a-Judge 1.5","oneLine":"A toolkit that converts robotic rollout videos into progress curves and computes metrics for failure progress, recovery, and execution quality, with a benchmark and evaluation suite for process reward models.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.14284","pdf":"https://arxiv.org/pdf/2608.14284","project":"https://prm-as-a-judge.github.io","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Moreover, we release a user-friendly assessment suite, including the benchmark, metric implementation, and visualization tools, to support reproducible manipulation process evaluation.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":14,"hfDailySubmittedAt":"2026-08-17T00:00:00.000Z","githubStars":32,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14284"},"ranking":{"30d":{"score":51,"rank":27,"coverage":0.85,"confidence":"High"},"90d":{"score":48,"rank":78,"coverage":0.7,"confidence":"Medium"}},"description":"A toolkit that converts robotic rollout videos into progress curves and computes metrics for failure progress, recovery, and execution quality, with a benchmark and evaluation suite for process reward models.","whyItMatters":"Supplies fine-grained process assessment beyond binary success, enabling more transparent and procedural evaluation of embodied models.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"e5906a6020650e486a8ca5b62cee08bba5bdeb2a1b876db31b7f2d7707a71259"},"motivation":"Fine-grained robotic evaluation matters for understanding embodied models, going beyond binary success rates and rule-based process scores.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The paper describes a toolkit with a released assessment suite including benchmark, metrics, and visualization tools, and provides a project page, establishing a public reuse path and stable scoring contract.","canonicalNameSource":"abstract","canonicalNameEvidence":"We present PRM-as-a-Judge 1.5, a toolkit for robot process assessment"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14284","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"The clear project page, toolkit release, and focus on reproducible process assessment for robotics are likely to drive interest from the embodied AI community."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_prmu_0126a6f3","familyId":"bmf_2cbd078f4d99","name":"PRMU","oneLine":"PRMU evaluates corpus-free multimodal unlearning of person-related knowledge in MLLMs, using textual and visual probes including adversarial evaluation and locality analysis.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.11149","pdf":"https://arxiv.org/pdf/2608.11149","project":null,"code":"https://github.com/2231122/PRMU","data":null,"hfPaper":"https://huggingface.co/papers/2608.11149"},"evidence":{"snippet":"To address this limitation, we introduce PRMU, a benchmark for evaluating corpus-free multimodal unlearning under realistic person-centric deletion requests.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11149"},"ranking":{"30d":{"score":23,"rank":153,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":357,"coverage":0.55,"confidence":"Low"}},"description":"PRMU evaluates corpus-free multimodal unlearning of person-related knowledge in MLLMs, using textual and visual probes including adversarial evaluation and locality analysis.","whyItMatters":"Addresses realistic deletion scenarios where original corpora are unavailable, providing a way to measure forgetting-locality trade-offs and vulnerability to knowledge reactivation, which is valuable for safe deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"99af0e1c783362a2968058a7c09d159952bec25f2c29ab41333c6380c8f8b792"},"motivation":"Multimodal large language models (MLLMs) have demonstrated remarkable capabilities in storing and recalling rich person-related knowledge, raising increasing concerns about reliable knowledge removal.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11149","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_proevent_18140068","familyId":"bmf_487414c0a9cc","name":"ProEvent","oneLine":"ProEvent is an event-centric benchmark for proactive agents, evaluating their ability to maintain a user's timetable from instant messaging chats. It assesses response timing, single-step correctness, and multi-step correctness using synthesized realistic chat scenarios.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.17701","pdf":"https://arxiv.org/pdf/2607.17701","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.17701"},"evidence":{"snippet":"To bridge these gaps, we introduce ProEvent, the first event-centric benchmark designed to assess an agent's ability to proactively maintain a user's timetable based on ongoing instant messaging chats.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.17701"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ProEvent is an event-centric benchmark for proactive agents, evaluating their ability to maintain a user's timetable from instant messaging chats. It assesses response timing, single-step correctness, and multi-step correctness using synthesized realistic chat scenarios.","whyItMatters":"The benchmark fills a gap in evaluating proactive agents for event-centric assistance, which is crucial for autonomous support. It provides practical value in measuring agents' ability to detect implicit events and reason from the user's perspective, revealing significant limitations in current models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1721fece9a4857fc0d2ef8370ed8fea2bbd1364392f26156e489f6dcae0781e4"},"motivation":"Proactive agents are expected to anticipate user needs and provide autonomous assistance by perceiving environmental context without explicit instructions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.17701","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_3afc7ee5d930fd7d","familyId":"catalog_family_3afc7ee5d930fd7d","name":"ProfBench","oneLine":"ProfBench evaluates models on professional-domain reasoning and knowledge-work tasks, including search-augmented question answering across expert fields.","description":"ProfBench evaluates models on professional-domain reasoning and knowledge-work tasks, including search-augmented question answering across expert fields.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/profbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3afc7ee5d930fd7d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/profbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"profbench","url":"https://llm-stats.com/benchmarks/profbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_long-horizon-forecasting-of-complete-finan_01c89589","familyId":"bmf_946f87e75fbb","name":"ProForma-20Q","oneLine":"Evaluates joint probabilistic forecasting of complete quarterly financial statements across 78 line items and 1-20 quarter horizons.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.11327","pdf":"https://arxiv.org/pdf/2608.11327","project":null,"code":"https://github.com/forma-lab-mccombs/proforma-20q","data":null,"hfPaper":null},"evidence":{"snippet":"We release ProForma-20Q, a reproducible benchmark for forecasting 78 statement line items 1-20 quarters ahead, for anonymized firms, from past statements and an industry code, scored by change-space $R^2$.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11327"},"ranking":{"30d":{"score":34,"rank":67,"coverage":0.55,"confidence":"Low"},"90d":{"score":33,"rank":215,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates joint probabilistic forecasting of complete quarterly financial statements across 78 line items and 1-20 quarter horizons.","whyItMatters":"Provides a reproducible financial forecasting benchmark that pushes beyond short-horizon, partial-statement evaluations used in prior work.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"84a2c3cf8da81de8249c4bf66118d1d051df03455f1b87460d782cb10295d84d"},"motivation":"Specialist training beats generalist scale when forecasting financial statements.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is explicitly named, has a public GitHub repository with a reproducible protocol, and defines a clear scoring CLI for submission.","canonicalNameSource":"abstract","canonicalNameEvidence":"We release ProForma-20Q, a reproducible benchmark for forecasting 78 statement line items 1-20 quarters ahead"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11327","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"Long-horizon financial statement forecasting is a technically demanding problem with a clear public benchmark, likely to interest the financial AI community."},"evaluationMode":"score_submission","publishers":[{"name":"Forma Lab, McCombs School of Business","organizationType":"academic-lab","sourceUrl":"https://github.com/forma-lab-mccombs/proforma-20q","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"lib_programbench","familyId":"family_programbench","name":"ProgramBench","oneLine":"Established benchmark family · Code & Software.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code & Software"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-05","releaseDatePrecision":"day","firstRelease":{"year":2026,"date":"2026-05-05"},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.03546","pdf":null,"project":"https://programbench.com/","code":"https://github.com/facebookresearch/ProgramBench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_programbench"},"ranking":{},"recordType":"family","aliases":["Program Bench"],"sourceAttribution":[{"role":"official-project","url":"https://programbench.com/"}],"adoptionRefs":[],"modelReportReferences":[],"catalogDiscoveryRefs":[],"catalogDiscoverySources":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"programBench","url":"https://benchlm.ai/benchmarks/programbench","paperUrl":"https://programbench.com/static/paper.pdf","year":"2026","fullName":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","format":"Cleanroom executable reimplementation","tasks":"200 program reconstruction tasks","successorKey":null},{"catalog":"benchlm","sourceId":"valsProgramBench","url":"https://benchlm.ai/benchmarks/valsprogrambench","paperUrl":"https://www.vals.ai/benchmarks/programbench","year":"2026","fullName":"Vals ProgramBench","format":"Accuracy score","tasks":"Program reconstruction tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"program-bench","url":"https://llm-stats.com/benchmarks/program-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","external","agents","code"],"catalogModelCount":6,"catalogStarCount":0},{"id":"catalog_2c8fb4e3c7fea53c","familyId":"catalog_family_2c8fb4e3c7fea53c","name":"ProgramBench (episode 1)","oneLine":"Program-reconstruction hidden-test pass rate after the first of five sequential long-context episodes.","description":"Program-reconstruction hidden-test pass rate after the first of five sequential long-context episodes.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2605.03546","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2c8fb4e3c7fea53c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/programbenchepisode1"}],"catalogSources":[{"catalog":"benchlm","sourceId":"programBenchEpisode1","url":"https://benchlm.ai/benchmarks/programbenchepisode1","paperUrl":"https://arxiv.org/abs/2605.03546","year":"2026","fullName":"ProgramBench hidden-test pass rate after episode 1","format":"Hidden-test pass rate after episode 1","tasks":"166 golden program-reconstruction tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_projectionbench_7fc1bdf2","familyId":"bmf_f6a58fd118f7","name":"ProjectionBench","oneLine":"A framework for evaluating scientific hypothesis generation in LLMs under progressive information disclosure, comparing model hypotheses to original paper conclusions via semantic similarity.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["AI Scientist","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30284","pdf":"https://arxiv.org/pdf/2605.30284","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30284"},"evidence":{"snippet":"We introduce a benchmark framework for evaluating model performance in scientific discovery and reasoning, building up from a raw problem to the classical null hypothesis test.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30284"},"ranking":{},"description":"A framework for evaluating scientific hypothesis generation in LLMs under progressive information disclosure, comparing model hypotheses to original paper conclusions via semantic similarity.","whyItMatters":"It assesses a model's innovativeness and grounded reasoning in scientific discovery, which is crucial for developing AI co-scientist systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"90fe5d9aab3c604f82a27c19214e253820e71f51ffa03ebb8fffa151ddbb52ce"},"motivation":"Scientific discovery is an inherently creative and uncertain process, requiring reasoning beyond the recall of known knowledge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30284","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_prompting-mammalps_0a70a842","familyId":"bmf_1eff7b878620","name":"Prompting-MammAlps","oneLine":"Prompting-MammAlps is a camera-trap text-to-video retrieval benchmark, evaluating video-language models on fine-grained retrieval of ecological events. It includes a test set of 135 queries and 775 candidate videos, with a proposed method that combines action localization with LLM-based parsing.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.09876","pdf":"https://arxiv.org/pdf/2607.09876","project":"https://cnai.epfl.ch/prompting-mammalps","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.09876"},"evidence":{"snippet":"In this work, we introduce Prompting-MammAlps, the first camera-trap TVR benchmark, and propose a fine-grained and interpretable TVR method.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.09876"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Prompting-MammAlps is a camera-trap text-to-video retrieval benchmark, evaluating video-language models on fine-grained retrieval of ecological events. It includes a test set of 135 queries and 775 candidate videos, with a proposed method that combines action localization with LLM-based parsing.","whyItMatters":"Text-to-video retrieval in ecological domains requires spatiotemporal understanding that current VLMs lack. This benchmark provides a standardized evaluation for fine-grained and interpretable retrieval, highlighting the limitations of zero-shot VLMs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b3dfdf6acd5bc7b3d4c219c330c950e703448df60f9ede43f16918f2edc71528"},"motivation":"Automatically retrieving videos from large camera-trap datasets remains challenging.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026","evidence":"Accepted at ECCV 2026; Project page: https://cnai.epfl.ch/prompting-mammalps","evidenceUrl":"https://arxiv.org/abs/2607.09876","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026","reviewStatus":"accepted","decisionRaw":"Accepted at ECCV 2026; Project page: https://cnai.epfl.ch/prompting-mammalps","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.09876","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at ECCV 2026; Project page: https://cnai.epfl.ch/prompting-mammalps","level":"author-claim"}]}],"publishers":[{"name":"EPFL CNAI","organizationType":"academic-lab","sourceUrl":"https://cnai.epfl.ch/prompting-mammalps","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_07f3f657a876386f","familyId":"catalog_family_07f3f657a876386f","name":"ProofBench","oneLine":"Vals AI automated theorem-proving benchmark.","description":"Vals AI automated theorem-proving benchmark.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/proof_bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_07f3f657a876386f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsproofbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsProofBench","url":"https://benchlm.ai/benchmarks/valsproofbench","paperUrl":"https://www.vals.ai/benchmarks/proof_bench","year":"2026","fullName":"Vals ProofBench","format":"Accuracy score","tasks":"Automated theorem proving","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_9b338560bbbf6210","familyId":"catalog_family_9b338560bbbf6210","name":"Protein Design","oneLine":"Generates novel protein sequences under family, topology, globularity, and structural-motif constraints.","description":"Generates novel protein sequences under family, topology, globularity, and structural-motif constraints.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9b338560bbbf6210"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/proteindesign"}],"catalogSources":[{"catalog":"benchlm","sourceId":"proteinDesign","url":"https://benchlm.ai/benchmarks/proteindesign","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Anthropic Protein Design evaluation","format":"Composite score","tasks":"Constrained protein-sequence design tasks","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_6318e5e8e172a2a8","familyId":"catalog_family_6318e5e8e172a2a8","name":"ProteinGym Hard","oneLine":"Predicts mutation effects by ranking mutant protein sequences against wild type and comparing against laboratory measurements.","description":"Predicts mutation effects by ranking mutant protein sequences against wild type and comparing against laboratory measurements.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6318e5e8e172a2a8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/proteingymhard"}],"catalogSources":[{"catalog":"benchlm","sourceId":"proteinGymHard","url":"https://benchlm.ai/benchmarks/proteingymhard","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"ProteinGym Hard","format":"Rank correlation","tasks":"Hard protein mutation-effect ranking tasks","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_b2dffdefb544c592","familyId":"catalog_family_b2dffdefb544c592","name":"ProtocolQA","oneLine":"ProtocolQA is a multiple-choice benchmark on troubleshooting failed experimental outcomes from common biological laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.","description":"ProtocolQA is a multiple-choice benchmark on troubleshooting failed experimental outcomes from common biological laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Safety","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/protocolqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b2dffdefb544c592"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/protocolqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"protocolqa","url":"https://llm-stats.com/benchmarks/protocolqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety","healthcare"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"catalog_fc8ebdd50fdeeddf","familyId":"catalog_family_fc8ebdd50fdeeddf","name":"Protocols (troubleshooting)","oneLine":"Detects and fixes errors in molecular-biology protocols using document, code, and web-search tools.","description":"Detects and fixes errors in molecular-biology protocols using document, code, and web-search tools.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fc8ebdd50fdeeddf"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/protocolstroubleshooting"}],"catalogSources":[{"catalog":"benchlm","sourceId":"protocolsTroubleshooting","url":"https://benchlm.ai/benchmarks/protocolstroubleshooting","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Molecular Biology Protocols Troubleshooting","format":"Task score","tasks":"Molecular-biology protocol troubleshooting","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_2dc3899a893c2ec0","familyId":"catalog_family_2dc3899a893c2ec0","name":"Protocols (understanding)","oneLine":"Extends online molecular-biology protocols in additional directions using document and web-search tools.","description":"Extends online molecular-biology protocols in additional directions using document and web-search tools.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2dc3899a893c2ec0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/protocolsunderstanding"}],"catalogSources":[{"catalog":"benchlm","sourceId":"protocolsUnderstanding","url":"https://benchlm.ai/benchmarks/protocolsunderstanding","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Benchling Molecular Biology Protocols Understanding","format":"Task score","tasks":"Molecular-biology protocol extension","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_protstructqa_a89c5890","familyId":"bmf_b5788fc59ce8","name":"ProtStructQA","oneLine":"ProtStructQA is an executable benchmark for protein structural question answering, with questions generated from DSL programs and answers obtained by executing on AlphaFold-predicted structures. Released 382.2K questions covering confidence, distances, PAE, solvent exposure, secondary structure, topology, and contacts.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00451","pdf":"https://arxiv.org/pdf/2606.00451","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.00451"},"evidence":{"snippet":"We introduce ProtStructQA, an executable benchmark for protein structural question answering in which each natural-language question is generated from a hidden typed domain-specific language (DSL) program and the answer is obtained by executing that program on an AlphaFold-predicted structure.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00451"},"ranking":{},"description":"ProtStructQA is an executable benchmark for protein structural question answering, with questions generated from DSL programs and answers obtained by executing on AlphaFold-predicted structures. Released 382.2K questions covering confidence, distances, PAE, solvent exposure, secondary structure, topology, and contacts.","whyItMatters":"Provides a diagnostic testbed for when language models can map words to executable 3D structural measurements, with a denotation threshold between model sizes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"73bb9d9469fde572881cefb5755315379cf673c5707beeabc0118ab93696644e"},"motivation":"Protein-language systems are often evaluated by whether they generate plausible biological text, but a structural question has a sharper semantics: it denotes a measurement in a 3D coordinate system.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00451","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_prweaver_3962c935","familyId":"bmf_064f6a3de339","name":"PRWeaver","oneLine":"PRWeaver evaluates LLM-based code auditors against malicious pull requests across repository history, using 208 attacks and multiple renderings.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02693","pdf":"https://arxiv.org/pdf/2608.02693","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.02693"},"evidence":{"snippet":"We introduce PRWeaver, a benchmark of 208 execution-validated attacks from ten real-world repositories, each instantiated under four matched review renderings (832 renderings in total).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02693"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PRWeaver evaluates LLM-based code auditors against malicious pull requests across repository history, using 208 attacks and multiple renderings.","whyItMatters":"Addresses reliability of code auditing agents under adversarial conditions, informing deployment of such systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a617830945e0073bc1f8afcd65e7f8d0d1b6fc97c53ab1af077a5f10f3d5adfe"},"motivation":"LLM-based code auditors are increasingly integrated into pull-request (PR) workflows, yet their reliability against adversarial changes distributed across repository evolution remains poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02693","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_pseudobench_085d3f1c","familyId":"bmf_9abe8256366d","name":"PseudoBench","oneLine":"PseudoBench evaluates agentic auto-research systems on their ability to identify and resist pseudoscientific narratives. It contains 200 curated pseudoscientific claim-evidence pairs across five domains and assesses performance through an end-to-end research pipeline from experiment design to report writing.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.18060","pdf":"https://arxiv.org/pdf/2606.18060","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18060"},"evidence":{"snippet":"We present PseudoBench, an adversarial benchmark for evaluating whether agentic auto-research systems can identify and resist pseudoscientific narratives.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18060"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PseudoBench evaluates agentic auto-research systems on their ability to identify and resist pseudoscientific narratives. It contains 200 curated pseudoscientific claim-evidence pairs across five domains and assesses performance through an end-to-end research pipeline from experiment design to report writing.","whyItMatters":"As AI agents increasingly participate in scientific research, their susceptibility to generating plausible but misleading studies poses a risk to academic integrity. PseudoBench provides a direct measure of this risk, offering a practical tool for assessing and improving agent safety before deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8ace3faed949f12e8f49b7f8e72cce3256c8a8f2e279991aa76c886a24dfc824"},"motivation":"As Large Language Model based agents enter autonomous scientific research, their ability to resist pseudoscience becomes increasingly important.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18060","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ptcg-bench_083265f8","familyId":"bmf_dc9b6ac4f245","name":"PTCG-Bench","oneLine":"PTCG-Bench evaluates LLM agents on the Pokémon Trading Card Game at two levels: single-environment decision-making and self-evolution through accumulated experience, with modular harness ablation to separate agent performance from harness design.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.29653","pdf":"https://arxiv.org/pdf/2605.29653","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29653"},"evidence":{"snippet":"We present PTCG-Bench, a benchmark built on the Pok'{e}mon Trading Card Game (PTCG) that evaluates LLM agents at two complementary levels: (1) their decision-making performance within a single complex environment, and (2) their ability to self-evolving through accumulated experience.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29653"},"ranking":{},"description":"PTCG-Bench evaluates LLM agents on the Pokémon Trading Card Game at two levels: single-environment decision-making and self-evolution through accumulated experience, with modular harness ablation to separate agent performance from harness design.","whyItMatters":"Agent benchmarks often miss strategic and evolving decision-making. PTCG-Bench provides a realistic interactive environment to assess self-evolution and harness sensitivity, which are critical for deploying autonomous agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9afca48c8eef6595edba7a7df5ce6d7d1f2ceddc2975d3bd7b7ee73adbb2d453"},"motivation":"Given a strategically complex board game, human players can quickly learn to devise strategies after playing a few rounds.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29653","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_ptxbench_11575102","familyId":"bmf_dc9322ec40cc","name":"PTXBench","oneLine":"Evaluates whether LLM agents can generate correct, architecture-specific GPU kernels that execute required PTX instructions and outperform libraries.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation","GPU kernel optimization","Architecture-specific programming","Iterative code repair"],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.17379","pdf":"https://arxiv.org/pdf/2608.17379","project":"https://github.com/zhang677/PTXBench","code":"https://github.com/zhang677/PTXBench","data":"https://huggingface.co/datasets/AccRL/accrl-training","hfPaper":"https://huggingface.co/papers/2608.17379"},"evidence":{"snippet":"We introduce PTXBench, a benchmark for evaluating and adapting large language models (LLMs) to use architecture-specific PTX for GPU kernel optimization.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":"2026-08-19T00:00:00.000Z","githubStars":14,"githubScope":"benchmark_repo","hfDatasetDownloads":112,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2608.17379"},"ranking":{"30d":{"score":37,"rank":57,"coverage":1.0,"confidence":"High","datasetDownloadRank":18,"datasetRankPopulation":30},"90d":{"score":36,"rank":191,"coverage":1.0,"confidence":"High","datasetDownloadRank":45,"datasetRankPopulation":66}},"description":"PTXBench evaluates LLMs in generating architecture-specific PTX for GPU kernel optimization. It measures functional correctness, execution of target instructions, and speedup over frontier libraries across GEMM and attention workloads on H100 and B200 GPUs.","whyItMatters":"Addresses the lack of standardized evaluation for LLM-driven GPU kernel optimization with architecture-specific PTX, providing a reproducible testbed to compare model capabilities and guide improvements in exploiting evolving GPU architectures.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"973bda730a80d543c43a5fee47f05df482a9a432407dfbc4a14ec612f974a73a"},"motivation":"We introduce PTXBench, a benchmark for evaluating and adapting large language models (LLMs) to use architecture-specific PTX for GPU kernel optimization.","constructionDetail":"PTXBench tests architecture-specific GPU kernel generation on H100 and B200 hardware, including correctness, required-instruction execution and speedup.","detail":{"taskBreakdown":["GEMM","Multi-head attention forward","Causal attention forward","Multi-head attention backward","Causal attention backward"],"protocol":{"tasks":"5 primary BF16 workloads plus generalization workloads","primaryMetric":"Correctness, target-instruction correctness, best speedup and Fast_p","language":"CUDA C++ with inline PTX","version":"arXiv v1"}},"curation":{"state":"source-reviewed","reviewedAt":"2026-08-20","sources":["https://arxiv.org/abs/2608.17379","https://arxiv.org/html/2608.17379","https://github.com/zhang677/PTXBench","https://huggingface.co/datasets/AccRL/accrl-training"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17379","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"score_submission","availability":{"paperStatus":"available","githubStatus":"available","hfDatasetStatus":"available","evaluatorStatus":"available","submissionStatus":"not_found"},"publishers":[{"name":"PTXBench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/zhang677/PTXBench","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_f8bc3d5d7b0f9193","familyId":"catalog_family_f8bc3d5d7b0f9193","name":"Public Benefits Bench v1","oneLine":"Can AI help people navigate SNAP benefits?","description":"Can AI help people navigate SNAP benefits?","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/public-benefits-bench-v1","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f8bc3d5d7b0f9193"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/publicbenefitsbenchv1"}],"catalogSources":[{"catalog":"benchlm","sourceId":"publicBenefitsBenchV1","url":"https://benchlm.ai/benchmarks/publicbenefitsbenchv1","paperUrl":"https://www.vals.ai/benchmarks/public-benefits-bench-v1","year":"2026","fullName":"Vals Public Benefits Bench v1","format":"Accuracy score","tasks":"SNAP public-benefits navigation tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_ce7110f730651068","familyId":"catalog_family_ce7110f730651068","name":"Public Benefits Bench v1.1","oneLine":"Can AI help people navigate SNAP benefits?","description":"Can AI help people navigate SNAP benefits?","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/public-benefits-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ce7110f730651068"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/publicbenefitsbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"publicBenefitsBench","url":"https://benchlm.ai/benchmarks/publicbenefitsbench","paperUrl":"https://www.vals.ai/benchmarks/public-benefits-bench","year":"2026","fullName":"Vals Public Benefits Bench v1.1","format":"Accuracy score","tasks":"SNAP public-benefits navigation tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_pyramathbench_14f9eba5","familyId":"bmf_560f968ae5dd","name":"PyraMathBench","oneLine":"PyraMathBench is a hierarchical benchmark with 32,505 questions derived from math word problems, evaluating numerical reasoning and mathematical capabilities across cognitive aspects and modalities.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03858","pdf":"https://arxiv.org/pdf/2606.03858","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03858"},"evidence":{"snippet":"We introduce PyraMathBench, a comprehensive hierarchical benchmark with 32,505 questions derived from 7,404 math word problems, spanning 4 key cognitive aspects, 14 subcategories, and 2 modalities.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03858"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"PyraMathBench is a hierarchical benchmark with 32,505 questions derived from math word problems, evaluating numerical reasoning and mathematical capabilities across cognitive aspects and modalities.","whyItMatters":"No public artifacts or reuse path are provided, and the benchmark lacks a clear scoring contract for independent runs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"de8d01ef9aab45b08c7c67582fd4ecabaa250fb9134f25d9d1a1907fd6bbd4d7"},"motivation":"Despite the pivotal role of numerical reasoning as the cornerstone of mathematical capabilities in large language models (LLMs) across applications, few benchmarks evaluate LLMs by integrating numerical processing and mathematical reasoning, hindering the interpretability of failures in math tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03858","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_a0984ed4cbce46aa","familyId":"catalog_family_a0984ed4cbce46aa","name":"Qasper","oneLine":"QASPER is a dataset of 5,049 information-seeking questions and answers anchored in 1,585 NLP research papers. Questions are written by NLP practitioners who read only titles and abstracts, while answers require understanding the full paper text and provide supporting evidence. The dataset challenges models with complex reasoning across document sections for academic document question answering. Each question seeks information present in the full text and is answered by a separate set of NLP practitioners who also provide supporting evidence to answers.","description":"QASPER is a dataset of 5,049 information-seeking questions and answers anchored in 1,585 NLP research papers. Questions are written by NLP practitioners who read only titles and abstracts, while answers require understanding the full paper text and provide supporting evidence. The dataset challenges models with complex reasoning across document sections for academic document question answering. Each question seeks information present in the full text and is answered by a separate set of NLP practitioners who also provide supporting evidence to answers.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/qasper","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a0984ed4cbce46aa"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/qasper"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"qasper","url":"https://llm-stats.com/benchmarks/qasper","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_4c4462e36bcf7053","familyId":"catalog_family_4c4462e36bcf7053","name":"QMSum","oneLine":"QMSum is a benchmark for query-based multi-domain meeting summarization consisting of 1,808 query-summary pairs over 232 meetings across academic, product, and committee domains. The dataset enables models to select and summarize relevant spans of meetings in response to specific queries. Published at NAACL 2021, QMSum presents significant challenges in long meeting summarization where models must identify and summarize relevant content based on user queries.","description":"QMSum is a benchmark for query-based multi-domain meeting summarization consisting of 1,808 query-summary pairs over 232 meetings across academic, product, and committee domains. The dataset enables models to select and summarize relevant spans of meetings in response to specific queries. Published at NAACL 2021, QMSum presents significant challenges in long meeting summarization where models must identify and summarize relevant content based on user queries.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Summarization"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/qmsum","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4c4462e36bcf7053"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/qmsum"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"qmsum","url":"https://llm-stats.com/benchmarks/qmsum","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","summarization"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_qo-bench_a8720b3c","familyId":"bmf_881e5bfa61b8","name":"QO-Bench","oneLine":"QO-Bench is a diagnostic benchmark for query-operator question answering over typed event tuples. It covers 22,984 news articles, 614 corporate events, and 18 query templates, with 785 questions. Gold answers are deterministically computed and scored by recall via exact match.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.04646","pdf":"https://arxiv.org/pdf/2606.04646","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04646"},"evidence":{"snippet":"We introduce QO-Bench, a diagnostic benchmark for query-operator question answering over typed event tuples.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04646"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"QO-Bench is a diagnostic benchmark for query-operator question answering over typed event tuples. It covers 22,984 news articles, 614 corporate events, and 18 query templates, with 785 questions. Gold answers are deterministically computed and scored by recall via exact match.","whyItMatters":"RAG systems may retrieve relevant passages but fail to preserve typed values needed for query operators. QO-Bench exposes this gap and allows operator-level diagnosis, guiding development of retrieval systems that preserve query semantics.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"48fb9c248162a794fe7dccec2e9caf8dbdf5ea3edd84a9cc951e9d33183517df"},"motivation":"Many real-world questions over business, legal, and scientific corpora are natural-language versions of database-style queries over records latent in text.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04646","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_quechuatok_31b7234f","familyId":"bmf_7dbc10d7377f","name":"QuechuaTok","oneLine":"A comparison of tokenization strategies (BPE, Unigram LM, WordPiece, PRPE) for Southern Quechua using a 200k-sentence corpus and a finite-state morphological analyzer as reference, with metrics including fertility rate, OOV rate, and morphological boundary accuracy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.23943","pdf":"https://arxiv.org/pdf/2606.23943","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.23943"},"evidence":{"snippet":"We present QuechuaTok, a systematic benchmark comparing four tokenization strategies - BPE, Unigram LM, WordPiece, and a morphology-aware PRPE tokenizer - for Southern Quechua (quz), a low-resource agglutinative language spoken by 8-10 million people in South America.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.23943"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A comparison of tokenization strategies (BPE, Unigram LM, WordPiece, PRPE) for Southern Quechua using a 200k-sentence corpus and a finite-state morphological analyzer as reference, with metrics including fertility rate, OOV rate, and morphological boundary accuracy.","whyItMatters":"Evaluates tokenizer quality for agglutinative low-resource languages, showing that fertility rate alone is insufficient and that morphological boundary accuracy provides a more meaningful signal.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7790df887552703ec0ab8f7d6a75a3d5ebd8533a63c2dccb304189ca55d56638"},"motivation":"Tokenization is a foundational step in NLP pipelines, yet standard evaluation metrics such as fertility rate fail to capture morphological correctness for agglutinative languages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.23943","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_quotebench_1e09ef6d","familyId":"bmf_04b51c8672a4","name":"QuoteBench","oneLine":"Benchmark of 56 one-shot tasks from 14 incident-derived families that measures exact final-state outcomes when LLM coding agents issue Bash commands through serializing/wrapping/reparsing transport, using an added unescaped parser.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":0.75,"links":{"report":"https://arxiv.org/abs/2608.13547","pdf":"https://arxiv.org/pdf/2608.13547","project":"https://quotebench.lsamc.website/","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"LLM coding agents issue Bash commands through interfaces that may serialize, wrap, and reparse model output.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":"2026-08-21T00:00:00.000Z","githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13547"},"ranking":{"30d":{"score":31,"rank":80,"coverage":0.85,"confidence":"High"},"90d":{"score":31,"rank":225,"coverage":0.7,"confidence":"Medium"}},"description":"Benchmark of 56 one-shot tasks from 14 incident-derived families that measures exact final-state outcomes when LLM coding agents issue Bash commands through serializing/wrapping/reparsing transport, using an added unescaped parser.","whyItMatters":"Reveals that matched execution scores can hide command-path failures, prompting evaluation of deployment configurations rather than treating model scores as intrinsic.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"917747c39f3e5a773f629fa8de0c9da92b90412ef0191fe746b1cbd219a394ca"},"motivation":"LLM coding agents issue Bash commands through interfaces that may serialize, wrap, and reparse model output.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The benchmark defines a repeatable evaluation object and scoring contract, with exact final-state validation and a 56-task frozen design featuring incident-derived families, and provides a public project link.","canonicalNameSource":"abstract","canonicalNameEvidence":"QuoteBench measures this boundary with exact final-state validation on 56 one-shot tasks"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13547","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses a critical failure mode in agent evaluation and has a dedicated project page, suggesting moderate initial interest."},"evaluationMode":"public_reusable","publishers":[{"name":"QuoteBench team","organizationType":"academic-lab","sourceUrl":"https://quotebench.lsamc.website/","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_qval_38cf43e7","familyId":"bmf_77df7a8b1bd4","name":"QVal","oneLine":"QVal is a training-free testbed for evaluating dense supervision signals for long-horizon LLM agents. It measures Q-alignment of scores from methods across four environments and seven families, without requiring training runs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.32034","pdf":"https://arxiv.org/pdf/2606.32034","project":null,"code":"https://github.com/bethgelab/qval","data":null,"hfPaper":"https://huggingface.co/papers/2606.32034"},"evidence":{"snippet":"We introduce QVal, a training-free testbed for directly evaluating dense supervision signals.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":12,"hfDailySubmittedAt":"2026-07-01T00:00:00.000Z","githubStars":8,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.32034"},"ranking":{"90d":{"score":38,"rank":164,"coverage":0.7,"confidence":"Medium"}},"description":"QVal is a training-free testbed for evaluating dense supervision signals for long-horizon LLM agents. It measures Q-alignment of scores from methods across four environments and seven families, without requiring training runs.","whyItMatters":"Fills the gap of direct, comparable evaluation of dense supervision methods, which is currently expensive and confounded. Enables early-stage assessment before training.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9cbee184baa666518cbe2891e72f220a38a9e4986850de93aedb917bf7ca30a5"},"motivation":"LLM agents increasingly act over long horizons, where a single trajectory can contain hundreds or thousands of actions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.32034","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"bethgelab","organizationType":"academic-lab","sourceUrl":"https://github.com/bethgelab/qval","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_b52a2c09f0a9ab89","familyId":"catalog_family_b52a2c09f0a9ab89","name":"QVHighlights","oneLine":"QVHighlights is a video moment retrieval benchmark for detecting moments and highlights in videos via natural language queries. Given a query, the model must localize the start and end times of relevant moments in the video, evaluated using metrics such as Recall@1 at a 0.5 IoU threshold.","description":"QVHighlights is a video moment retrieval benchmark for detecting moments and highlights in videos via natural language queries. Given a query, the model must localize the start and end times of relevant moments in the video, evaluated using metrics such as Recall@1 at a 0.5 IoU threshold.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/qvhighlights","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b52a2c09f0a9ab89"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/qvhighlights"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"qvhighlights","url":"https://llm-stats.com/benchmarks/qvhighlights","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","video","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_qwen-image-bench_7fd96584","familyId":"bmf_6aa256ab0afc","name":"Qwen-Image-Bench","oneLine":"Qwen-Image-Bench evaluates text-to-image models on five pillars including Real-world Fidelity and Creative Generation, with 1000 prompts and 56 rubric-based facets scored by a trained judge model.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.28091","pdf":"https://arxiv.org/pdf/2605.28091","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28091"},"evidence":{"snippet":"To address the gap, we introduce Qwen-Image-Bench, a creator-centric benchmark co-designed with professional artists and grounded in real-world creation scenarios.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28091"},"ranking":{},"description":"Qwen-Image-Bench evaluates text-to-image models on five pillars including Real-world Fidelity and Creative Generation, with 1000 prompts and 56 rubric-based facets scored by a trained judge model.","whyItMatters":"It aims to address gaps in existing T2I benchmarks by assessing application-driven capabilities for professional creative workflows, offering fine-grained diagnostics for model comparison and development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ccd3df3cfb163ea85404c59d4466f421e5f52fc8246bf5469ed60dc68adbeefa"},"motivation":"Text-to-Image generation has evolved from basic image synthesis into a frequently used core capability in professional creative workflows, where simple text-image alignment can no longer satisfy users' pressing demands for faithful real-world reconstruction and genuine creative expression.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28091","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_qwen3-8-27b-inference-benchmark-4090_328340c0","familyId":"bmf_c8f88fa92b7e","name":"qwen3.8-27b-inference-benchmark-4090","oneLine":"The dataset presents performance and accuracy results for specific serving configurations, but does not define a reusable evaluation protocol or scoring contract for other teams.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Inspectable","releasedAt":"2026-08-21","firstSeenAt":"2026-08-22","recognitionConfidence":0.4,"links":{"report":"https://huggingface.co/datasets/pxzleo/qwen3.8-27b-inference-benchmark-4090","pdf":null,"project":null,"code":null,"data":"https://huggingface.co/datasets/pxzleo/qwen3.8-27b-inference-benchmark-4090","hfPaper":null},"evidence":{"snippet":"Qwen3.8-27B Inference Benchmark on RTX 4090 48GB 中文说明 · GitHub benchmark repository Structured performance and accuracy results for four real Qwen3.8-27B serving configurations on an NVIDIA RTX 4090 48 GB workstation.","reasonCodes":["discovered via huggingface","benchmark term in abstract","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":96,"hfDatasetLikes":1},"source":{"type":"huggingface","id":"huggingface:pxzleo/qwen3.8-27b-inference-benchmark-4090"},"ranking":{"30d":{"score":50,"rank":29,"coverage":0.15,"confidence":"Low","datasetDownloadRank":19,"datasetRankPopulation":30},"90d":{"score":46,"rank":92,"coverage":0.3,"confidence":"Low","datasetDownloadRank":46,"datasetRankPopulation":66}},"description":"The dataset presents performance and accuracy results for specific serving configurations, but does not define a reusable evaluation protocol or scoring contract for other teams.","whyItMatters":"The results are hardware-specific and do not offer a standardized way to compare models beyond the recorded deployments.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"d283d96e08dd623fe4634a8bc59b4bede55e53b80f9c0ea91a81044573082e4f"},"motivation":"Qwen3.8-27B Inference Benchmark on RTX 4090 48GB 中文说明 · GitHub benchmark repository Structured performance and accuracy results for four real Qwen3.8-27B serving configurations on an NVIDIA RTX 4090 48 GB workstation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The Hugging Face dataset page is a repository of results and metadata without an independent paper or formal benchmark declaration, so it is deferred pending clearer evidence."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://huggingface.co/datasets/pxzleo/qwen3.8-27b-inference-benchmark-4090","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-22T19:19:35.289219Z"},"attentionForecast":{"score":0,"confidence":"Low","horizon":"7d","reason":"Excluded or deferred due to lack of formal benchmark release."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_c6600ab2bf5eefa7","familyId":"catalog_family_c6600ab2bf5eefa7","name":"QwenClawBench","oneLine":"QwenClawBench is a real-user-distribution Claw agent benchmark for evaluating coding agents on realistic developer tasks.","description":"QwenClawBench is a real-user-distribution Claw agent benchmark for evaluating coding agents on realistic developer tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c6600ab2bf5eefa7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/qwenclawbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/qwenclawbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"qwenClawBench","url":"https://benchlm.ai/benchmarks/qwenclawbench","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"QwenClawBench","format":"End-to-end agent evaluation","tasks":"Real-world agent workflows","successorKey":null},{"catalog":"llm-stats","sourceId":"qwenclawbench","url":"https://llm-stats.com/benchmarks/qwenclawbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_5590bdccdada7e0a","familyId":"catalog_family_5590bdccdada7e0a","name":"QwenQoderBench","oneLine":"QwenQoderBench is Qwen's internal benchmark for evaluating coding-agent performance.","description":"QwenQoderBench is Qwen's internal benchmark for evaluating coding-agent performance.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/qwen-qoder-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5590bdccdada7e0a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/qwen-qoder-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"qwen-qoder-bench","url":"https://llm-stats.com/benchmarks/qwen-qoder-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_0cc3a79c633bc816","familyId":"catalog_family_0cc3a79c633bc816","name":"QwenReactBench","oneLine":"QwenReactBench is Qwen's internal React application generation benchmark, reported as a BT/Elo rating.","description":"QwenReactBench is Qwen's internal React application generation benchmark, reported as a BT/Elo rating.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Multimodal","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.8","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0cc3a79c633bc816"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/qwenreactbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/qwen-react-bench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"qwenReactBench","url":"https://benchlm.ai/benchmarks/qwenreactbench","paperUrl":"https://qwen.ai/blog?id=qwen3.8","year":"2026","fullName":"QwenReactBench","format":"Bradley-Terry/Elo rating","tasks":"Bilingual React project construction","successorKey":null},{"catalog":"llm-stats","sourceId":"qwen-react-bench","url":"https://llm-stats.com/benchmarks/qwen-react-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","multimodal","agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_4cc2ff3b00c610fe","familyId":"catalog_family_4cc2ff3b00c610fe","name":"QwenSVG","oneLine":"QwenSVG is Qwen's internal SVG generation benchmark for evaluating front-end and visual code generation. Scores are reported as BT/Elo ratings from auto-rendered outputs judged by a multimodal evaluator.","description":"QwenSVG is Qwen's internal SVG generation benchmark for evaluating front-end and visual code generation. Scores are reported as BT/Elo ratings from auto-rendered outputs judged by a multimodal evaluator.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/qwen-svg","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4cc2ff3b00c610fe"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/qwen-svg"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"qwen-svg","url":"https://llm-stats.com/benchmarks/qwen-svg","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","agents","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_e9379882dbaa5cbc","familyId":"catalog_family_e9379882dbaa5cbc","name":"QwenSWEBench","oneLine":"QwenSWEBench is Qwen's software-engineering agent benchmark for repository-level issue resolution.","description":"QwenSWEBench is Qwen's software-engineering agent benchmark for repository-level issue resolution.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/qwen-swe-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e9379882dbaa5cbc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/qwen-swe-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"qwen-swe-bench","url":"https://llm-stats.com/benchmarks/qwen-swe-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_0ea8424382098997","familyId":"catalog_family_0ea8424382098997","name":"QwenWebBench","oneLine":"QwenWebBench is an internal front-end code generation benchmark by Qwen. It is bilingual (EN/CN) and spans 7 categories (Web Design, Web Apps, Games, SVG, Data Visualization, Animation, and 3D), using auto-render plus a multimodal judge for code and visual correctness. Scores are reported as BT/Elo ratings.","description":"QwenWebBench is an internal front-end code generation benchmark by Qwen. It is bilingual (EN/CN) and spans 7 categories (Web Design, Web Apps, Games, SVG, Data Visualization, Animation, and 3D), using auto-render plus a multimodal judge for code and visual correctness. Scores are reported as BT/Elo ratings.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Multimodal","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0ea8424382098997"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/qwenwebbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/qwenwebbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"qwenWebBench","url":"https://benchlm.ai/benchmarks/qwenwebbench","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"QwenWebBench","format":"Elo-style artifact benchmark","tasks":"Web artifacts and interactive deliverables","successorKey":null},{"catalog":"llm-stats","sourceId":"qwenwebbench","url":"https://llm-stats.com/benchmarks/qwenwebbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","multimodal","agents","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_790e679631e4dd86","familyId":"catalog_family_790e679631e4dd86","name":"QwenWorldBench","oneLine":"QwenWorldBench is Qwen's internal benchmark for evaluating LLMs as world models that simulate agentic environments across Terminal, SWE, MCP, Search, OS, Android, and Web domains.","description":"QwenWorldBench is Qwen's internal benchmark for evaluating LLMs as world models that simulate agentic environments across Terminal, SWE, MCP, Search, OS, Android, and Web domains.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Simulation","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/qwenworldbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_790e679631e4dd86"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/qwenworldbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"qwenworldbench","url":"https://llm-stats.com/benchmarks/qwenworldbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","simulation","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_r2handoversim_321897f0","familyId":"bmf_bd02714c5c81","name":"R2HandoverSim","oneLine":"Simulation benchmark for robot-to-human object handovers with standardized evaluation protocol and five complementary metrics: planning feasibility, reachability, grasp stability, grasp affordance, and safety. Compares four baseline methods on shared grasp pose prediction.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-19","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.21011","pdf":"https://arxiv.org/pdf/2606.21011","project":"https://robot-future.github.io/r2handoversim/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.21011"},"evidence":{"snippet":"We present R2HandoverSim, a simulation benchmark for robot-to-human (R2H) object handovers.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.21011"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Simulation benchmark for robot-to-human object handovers with standardized evaluation protocol and five complementary metrics: planning feasibility, reachability, grasp stability, grasp affordance, and safety. Compares four baseline methods on shared grasp pose prediction.","whyItMatters":"Addresses lack of standardized evaluation in R2H handover research, enabling objective comparison across methods. Simulation results correlate with real-world outcomes, providing a practical tool for developing and validating handover systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"69b9be08f981b88f649546214bc3f7bba2709a8a5b79a71e8d6f33324de7aa62"},"motivation":"We present R2HandoverSim, a simulation benchmark for robot-to-human (R2H) object handovers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)","evidence":"Accepted by the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)","evidenceUrl":"https://arxiv.org/abs/2606.21011","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)","reviewStatus":"accepted","decisionRaw":"Accepted by the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.21011","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by the 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)","level":"author-claim"}]}],"publishers":[{"name":"Robot Future","organizationType":"community","sourceUrl":"https://robot-future.github.io/r2handoversim/","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_r2m-bench_1ea8e143","familyId":"bmf_4bb72cc2e03a","name":"R2M-Bench","oneLine":"High similarity between first-visit and return frames does not necessarily show that a video world model remembered the scene; the intervening rollout may simply have changed very little.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.27328","pdf":"https://arxiv.org/pdf/2608.27328","project":null,"code":"https://github.com/AMAP-ML/R2MBench","data":null,"hfPaper":"https://huggingface.co/papers/2608.27328"},"evidence":{"snippet":"We introduce \\emph{R2M-Bench} (\\textbf{R}elative \\textbf{R}evisit \\textbf{M}emory Benchmark), a benchmark of observable revisit-selective consistency.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27328"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"High similarity between first-visit and return frames does not necessarily show that a video world model remembered the scene; the intervening rollout may simply have changed very little.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.27328","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_r-3-bench-llms-struggle-with-resource-rati_fbfbc38c","familyId":"bmf_6d88db5c68c9","name":"R3-Bench","oneLine":"Evaluates resource-rational reasoning by measuring how models allocate a shared budget across six-problem suites in mathematics, competitive programming, and abstract reasoning, with scoring based on completed problems.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Code generation"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.16033","pdf":"https://arxiv.org/pdf/2608.16033","project":null,"code":"https://github.com/NineAbyss/R-3-Bench","data":"https://huggingface.co/datasets/R-3-Bench/R-3-Bench","hfPaper":null},"evidence":{"snippet":"We introduce $R^3$-Bench, which evaluates six-problem suites under shared budgets across mathematics, competitive programming, and abstract reasoning in tool-free and agentic settings.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":12,"hfDailySubmittedAt":"2026-08-18T00:00:00.000Z","githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":413,"hfDatasetLikes":1},"source":{"type":"arxiv","id":"2608.16033"},"ranking":{"30d":{"score":32,"rank":70,"coverage":1.0,"confidence":"High","datasetDownloadRank":9,"datasetRankPopulation":30},"90d":{"score":30,"rank":241,"coverage":1.0,"confidence":"High","datasetDownloadRank":24,"datasetRankPopulation":66}},"description":"Evaluates resource-rational reasoning by measuring how models allocate a shared budget across six-problem suites in mathematics, competitive programming, and abstract reasoning, with scoring based on completed problems.","whyItMatters":"Addresses the gap between demonstrated single-problem competence and shared-budget allocation, providing a fixed dataset and protocol for studying resource allocation in LLMs.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"2d149a6a33a106980178af7ce27e01bd513c59b5924dcff4b418fcc409a5aa12"},"motivation":"In cognitive science, resource rationality asks how an agent should allocate limited computation to maximize expected value.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"R3-Bench has a formal name, a stable scoring contract documented in the evaluator, and public code and data on GitHub and Hugging Face enabling reuse and comparison.","canonicalNameSource":"paper_title","canonicalNameEvidence":"$R^3$-Bench: LLMs Struggle with Resource-Rational Reasoning under Shared Budgets"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.16033","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":55,"confidence":"Medium","reason":"Clear public artifacts and a novel resource-rational framing are likely to draw moderate interest from reasoning and agent evaluation researchers.","horizon":"7d"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_ra-bench_8deff463","familyId":"bmf_d6d2987131f5","name":"RA-Bench","oneLine":"RA-Bench evaluates AI-generated video detection using real crisis-event videos as anchors. It includes 17,886 videos: 1,830 real anchors across 10 social-risk categories and 16,056 generated clips from nine generators. Detection is scored per source via AUC and TPR@5%FPR, with separate tracks for human-deceptive and dissemination-processed videos.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.14391","pdf":"https://arxiv.org/pdf/2608.14391","project":null,"code":"https://github.com/24029100313/RA-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.14391"},"evidence":{"snippet":"To address this gap, we introduce RA-Bench, a benchmark for AI-generated video detection that uses Real videos as Anchors.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":280,"hfDailySubmittedAt":"2026-08-17T00:00:00.000Z","githubStars":94,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14391"},"ranking":{"30d":{"score":75,"rank":2,"coverage":0.85,"confidence":"High"},"90d":{"score":64,"rank":14,"coverage":0.7,"confidence":"Medium"}},"description":"RA-Bench evaluates AI-generated video detection using real crisis-event videos as anchors. It includes 17,886 videos: 1,830 real anchors across 10 social-risk categories and 16,056 generated clips from nine generators. Detection is scored per source via AUC and TPR@5%FPR, with separate tracks for human-deceptive and dissemination-processed videos.","whyItMatters":"Current detectors lack evidence on realistic AI-generated crisis videos. RA-Bench provides a matched real-generated evaluation to assess detector generalization, human perception, and robustness to social dissemination, supporting practical decisions on misinformation defense.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"856c736b7dc8596472120f97a8e1b966a697ea59dc18f7f66978cc093746679d"},"motivation":"Recent video generators can fabricate realistic depictions of wars, disasters, public emergencies, and other real-world crises, creating substantial risks of misinformation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14391","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_rag-retrieval-benchmark_2e75b7eb","familyId":"bmf_033fb9287072","name":"rag-retrieval-benchmark","oneLine":"The repository compares seven retrieval configurations on FiQA queries using standard retrieval metrics, but does not self-identify as a named benchmark release.","area":"Language & Knowledge","applicationDomains":["Cybersecurity","Finance & Economics"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity","Financial Services"],"capabilities":["Information retrieval"],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-23","firstSeenAt":"2026-08-24","recognitionConfidence":0.4,"links":{"report":"https://github.com/gauthamRohan/rag-retrieval-benchmark","pdf":null,"project":null,"code":"https://github.com/gauthamRohan/rag-retrieval-benchmark","data":"https://huggingface.co/datasets/BeIR/fiqa","hfPaper":null},"evidence":{"snippet":"rag-retrieval-benchmark Does hybrid search and reranking actually help?","reasonCodes":["discovered via github","benchmark term in abstract","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":3205,"hfDatasetLikes":16},"source":{"type":"github","id":"github:gauthamrohan/rag-retrieval-benchmark"},"ranking":{"30d":{"score":28,"rank":99,"coverage":0.7,"confidence":"Medium","datasetDownloadRank":2,"datasetRankPopulation":30},"90d":{"score":28,"rank":285,"coverage":0.85,"confidence":"High","datasetDownloadRank":7,"datasetRankPopulation":66}},"description":"The repository compares seven retrieval configurations on FiQA queries using standard retrieval metrics, but does not self-identify as a named benchmark release.","whyItMatters":"The measured tradeoffs between dense retrieval, BM25, and reranking could inform retrieval system design, but the artifact lacks a formal benchmark declaration.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"50fa49ac2a9ebbd043f1036c8ce7530a20ddc200aa0c868bdb5e1708400bb1c4"},"motivation":"rag-retrieval-benchmark Does hybrid search and reranking actually help?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The GitHub repository presents an analysis of retrieval strategies without a formally named benchmark, independent paper, or explicit public comparison protocol."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/gauthamRohan/rag-retrieval-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-24T07:04:32.475961Z"},"attentionForecast":{"score":30,"confidence":"Low","horizon":"7d","reason":"The repository provides benchmark-like results but lacks clear benchmark naming and independent validation, limiting its expected reach."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"cross-domain"},{"id":"bm_randombench_dfec6d0b","familyId":"bmf_6cb8c3fc76bb","name":"RandomBench","oneLine":"RandomBench evaluates whether multimodal LLMs maintain distributionally neutral behavior when selecting among equivalent options, providing metrics for entropy and distributional bias under explicit random instructions.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05874","pdf":"https://arxiv.org/pdf/2606.05874","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05874"},"evidence":{"snippet":"To bridge this gap, we propose RandomBench, a benchmark designed to evaluate whether MLLMs can maintain distributionally neutral behavior when selecting among equivalent options.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05874"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RandomBench evaluates whether multimodal LLMs maintain distributionally neutral behavior when selecting among equivalent options, providing metrics for entropy and distributional bias under explicit random instructions.","whyItMatters":"Logic-neutral scenarios are underexplored in MLLM evaluation. RandomBench introduces a way to quantify stochastic collapse, a bias toward non-uniform choices that affects repetitive behavior and coverage, aiding design of more robust models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"deb5a31edd9729ff7bcf690a98368458e859d96d602081a530e7924359a79475"},"motivation":"Current evaluations for Multimodal Large Language Models (MLLMs) overwhelmingly focus on utility-driven objectives, leaving model behavior under logic-neutral scenarios largely underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05874","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_ratio-a-benchmark-for-retrieval-across-typ_4be3618a","familyId":"bmf_08215d044ab0","name":"RATIO","oneLine":"Evaluates retrieval models on three ideation moves—Address, Broaden, Specify—using relevance judgments derived from full-text scientific papers across computer science literature.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":0.45,"links":{"report":"https://arxiv.org/abs/2608.27394","pdf":"https://arxiv.org/pdf/2608.27394","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce RATIO (Retrieval Across Typed Ideation Operations), a large-scale benchmark in which relevance is defined by three operations which we name ideation moves: Address retrieves potential approaches for stated problems, Broaden retrieves more general formulations, and Specify retrieves concrete instantiations.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":null,"hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27394"},"ranking":{},"description":"Evaluates retrieval models on three ideation moves—Address, Broaden, Specify—using relevance judgments derived from full-text scientific papers across computer science literature.","whyItMatters":"Defines relevance by operations that support literature-grounded ideation, enabling training and evaluation of retrieval systems tailored to scientific inspiration.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T07:23:36.749451Z","inputHash":"2740e80d7ded8a1da3da3f9c8766e8aecb8cf055ea47942f41cf9406b24d9f07"},"motivation":"Retrieved scientific literature can serve as inspiration for both human and AI scientists.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T07:23:36.749451Z","model":"deepseek-v4-pro","decisionReason":"The paper formally introduces RATIO as a benchmark with defined moves, construction recipe, and experiments; it provides a scalable framework intended for ongoing model comparison. No direct artifact link is supplied, but the abstract indicates a public benchmark with stable evaluation.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce RATIO (Retrieval Across Typed Ideation Operations), a large-scale benchmark"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.27394","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-30T06:59:55.242506Z"},"attentionForecast":{"score":59,"confidence":"Low","horizon":"7d","reason":"The benchmark spans multiple ideation moves and is built from millions of papers, which may attract attention from retrieval and scientific NLP communities despite lacking a direct artifact link."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_rdvsv2_d98f1ace","familyId":"bmf_57f7f6186a02","name":"RDVSv2","oneLine":"RDVSv2 is a large-scale benchmark for RGB-D video salient object detection, containing 249 video sequences with 29,077 annotated frames. It includes depth maps, optical flow, and eye-tracking-guided salient object masks. The benchmark provides a fixed dataset and evaluation protocol for comparing models on this task.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25392","pdf":"https://arxiv.org/pdf/2607.25392","project":null,"code":"https://github.com/ltynick/RDVSv2","data":null,"hfPaper":"https://huggingface.co/papers/2607.25392"},"evidence":{"snippet":"We introduce RDVSv2, a large-scale benchmark for RGB-D video salient object detection (RGB-D VSOD) with dense frame-level annotations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25392"},"ranking":{"90d":{"score":35,"rank":199,"coverage":0.55,"confidence":"Low"}},"description":"RDVSv2 is a large-scale benchmark for RGB-D video salient object detection, containing 249 video sequences with 29,077 annotated frames. It includes depth maps, optical flow, and eye-tracking-guided salient object masks. The benchmark provides a fixed dataset and evaluation protocol for comparing models on this task.","whyItMatters":"Existing RGB-D VSOD datasets are limited in scale and annotation quality, hindering progress. RDVSv2 offers a larger, more diverse, and challenging benchmark, enabling more robust evaluation and comparison of models, and supporting the development of methods that can handle real-world scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"64e9f1b0c825d0ba6a0fdb39fb41668e0465a2950a36f3a4309d7e26da23db98"},"motivation":"We introduce RDVSv2, a large-scale benchmark for RGB-D video salient object detection (RGB-D VSOD) with dense frame-level annotations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ACMMM 2026","evidence":"Accepted to ACMMM 2026","evidenceUrl":"https://arxiv.org/abs/2607.25392","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ACMMM 2026","reviewStatus":"accepted","decisionRaw":"Accepted to ACMMM 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.25392","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to ACMMM 2026","level":"author-claim"}]}],"publishers":[{"name":"RDVSv2 team","organizationType":"academic-lab","sourceUrl":"https://github.com/ltynick/RDVSv2","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_104856ca1483e73f","familyId":"catalog_family_104856ca1483e73f","name":"React Native Evals","oneLine":"An open benchmark for AI coding agents on real-world React Native implementation tasks, emphasizing working app behavior, recommended architecture choices, and strict constraint adherence.","description":"An open benchmark for AI coding agents on real-world React Native implementation tasks, emphasizing working app behavior, recommended architecture choices, and strict constraint adherence.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://rn-evals.vercel.app/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_104856ca1483e73f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/reactnativeevals"}],"catalogSources":[{"catalog":"benchlm","sourceId":"reactNativeEvals","url":"https://benchlm.ai/benchmarks/reactnativeevals","paperUrl":"https://rn-evals.vercel.app/","year":"2026","fullName":"React Native Evals","format":"Framework-specific app development evaluation","tasks":"React Native app implementation tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_reactbench_423fd886","familyId":"bmf_267a0e871f29","name":"ReactBench","oneLine":"ReactBench evaluates multimodal large language models on cause-driven hallucination through four targeted tasks (Relational Erasure, Counterfactual Attribute, Alteration Tracing, Dense Counting) with exam-style evaluation and chain-of-thought reasoning for sub-cause identification.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29579","pdf":"https://arxiv.org/pdf/2605.29579","project":"https://reactbench.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29579"},"evidence":{"snippet":"To address these limitations, we introduce ReactBench, a cause-driven hallucination benchmark featuring multiple tasks and an exam-style evaluation format.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29579"},"ranking":{},"description":"ReactBench evaluates multimodal large language models on cause-driven hallucination through four targeted tasks (Relational Erasure, Counterfactual Attribute, Alteration Tracing, Dense Counting) with exam-style evaluation and chain-of-thought reasoning for sub-cause identification.","whyItMatters":"Existing hallucination benchmarks measure outcomes rather than causes. ReactBench provides a systematic testbed to diagnose specific failure modes like co-occurrence bias and fine-grained perceptual bottlenecks, offering interpretable insights for model robustness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"04235a5976e036e410911eb527237f9483bd5788631f34baffad5d62291a5fd4"},"motivation":"While multimodal large language models (MLLMs) have achieved rapid progress in vision-language understanding, they remain prone to multimodal hallucinations, producing responses that are inconsistent with the visual input.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29579","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ReactBench Project","organizationType":"community","sourceUrl":"https://reactbench.github.io/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"reactBench","url":"https://benchlm.ai/benchmarks/reactbench","paperUrl":"https://www.reactbench.com/","year":"2026","fullName":"ReactBench v1","format":"Pass@1 weighted rubric score","tasks":"51 production React tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0},{"id":"bm_reactsim-bench_5401cd81","familyId":"bmf_a6b25ed93a48","name":"ReactSim-Bench","oneLine":"ReactSim-Bench evaluates reactive capability of behavior world model simulators in autonomous driving by decoupling agent and AV control, using collision, map, and kinematic metrics on 2,636 scenarios.","area":"Robotics & Embodied AI","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Automotive"],"capabilities":[],"topics":["cs.RO"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-12","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.14058","pdf":"https://arxiv.org/pdf/2606.14058","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.14058"},"evidence":{"snippet":"In this work, we introduce ReactSim-Bench for evaluating the reactive capability of behavior world model simulation in autonomous driving.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.14058"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ReactSim-Bench evaluates reactive capability of behavior world model simulators in autonomous driving by decoupling agent and AV control, using collision, map, and kinematic metrics on 2,636 scenarios.","whyItMatters":"Existing sim benchmarks don't directly measure whether simulated agents respond feasibly to novel AV behaviors; this benchmark fills that gap for safe simulation-based testing.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9dd59cb24a0c58188de4590d7f9edced5b30bda4b5d859f8ff45e157eb2da664"},"motivation":"Reactive capability is a key property of data-driven behavior world model simulators for autonomous driving simulation systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.14058","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_real-bench_906823b4","familyId":"bmf_f0848f948867","name":"REAL-Bench","oneLine":"REAL-Bench evaluates vision-driven embodied agents in open-world mobile manipulation across 241 tasks spanning active exploration, visual distraction, articulated manipulation, and interactive disambiguation. The benchmark provides standardized task definitions and a simulator-based evaluation protocol.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.13653","pdf":"https://arxiv.org/pdf/2607.13653","project":null,"code":"https://github.com/InternRobotics/REAL","data":null,"hfPaper":"https://huggingface.co/papers/2607.13653"},"evidence":{"snippet":"To comprehensively evaluate this approach, we introduce REAL-Bench, a benchmark spanning 241 tasks across active exploration, visual distraction, articulated manipulation, and interactive disambiguation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":40,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.13653"},"ranking":{"90d":{"score":43,"rank":129,"coverage":0.7,"confidence":"Medium"}},"description":"REAL-Bench evaluates vision-driven embodied agents in open-world mobile manipulation across 241 tasks spanning active exploration, visual distraction, articulated manipulation, and interactive disambiguation. The benchmark provides standardized task definitions and a simulator-based evaluation protocol.","whyItMatters":"The benchmark targets the gap between simulation and real-world deployment for embodied agents, offering a repeatable evaluation for long-horizon tasks requiring visual grounding and interactive intent disambiguation. It enables systematic comparison of agent frameworks and informs progress toward practical deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d87a8843dd9c4628b22e06c66c59b9940846649238afd448749ac5ba5b990da7"},"motivation":"Real-world deployment of embodied agents requires active exploration, visual grounding, and interactive intent disambiguation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026","evidence":"Accepted to ECCV 2026. 57 pages. Code available at https://github.com/InternRobotics/REAL","evidenceUrl":"https://arxiv.org/abs/2607.13653","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026","reviewStatus":"accepted","decisionRaw":"Accepted to ECCV 2026. 57 pages. Code available at https://github.com/InternRobotics/REAL","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.13653","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to ECCV 2026. 57 pages. Code available at https://github.com/InternRobotics/REAL","level":"author-claim"}]}],"publishers":[{"name":"InternRobotics","organizationType":"academic-lab","sourceUrl":"https://github.com/InternRobotics/REAL","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_realbench_3d168523","familyId":"bmf_f0848f948867","name":"RealBench","oneLine":"Evaluates data-driven numerical weather forecasting models under operational conditions, using strictly out-of-distribution test data from 2025 and integrating low-latency operational analysis and large-scale in-situ observations from over 10,000 stations. It provides metrics for global forecasting and for extreme events such as heatwaves, cold surges, and tropical cyclones.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24945","pdf":"https://arxiv.org/pdf/2605.24945","project":null,"code":"https://github.com/lixruize-del/NWP-Benchmark","data":null,"hfPaper":"https://huggingface.co/papers/2605.24945"},"evidence":{"snippet":"In this work, we introduce RealBench, a next-generation benchmark for AI weather forecasting that emphasizes realistic evaluation under operational conditions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":7,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24945"},"ranking":{},"description":"Evaluates data-driven numerical weather forecasting models under operational conditions, using strictly out-of-distribution test data from 2025 and integrating low-latency operational analysis and large-scale in-situ observations from over 10,000 stations. It provides metrics for global forecasting and for extreme events such as heatwaves, cold surges, and tropical cyclones.","whyItMatters":"Existing benchmarks rely on reanalysis products that do not reflect real-time operational constraints, leading to mismatches between benchmark scores and real-world performance. This benchmark provides a more faithful and operationally relevant evaluation paradigm.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"09b5679de8cfa25c72850359b6871f73ef98c3bab38d223f8171a79b473b10c5"},"motivation":"Accurate evaluation of weather forecasting models is critical for their reliable deployment in real-world applications.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24945","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"NWP-Benchmark contributors","organizationType":"academic-lab","sourceUrl":"https://github.com/lixruize-del/NWP-Benchmark","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_realdesed_918e47e6","familyId":"bmf_219e72d356ca","name":"RealDESED","oneLine":"RealDESED is a real-world domestic sound event detection benchmark with 5,710 recordings from 652 participants, 15 classes, and temporally precise annotations. Includes multi-annotator labeling and rich metadata.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["eess.AS"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.16736","pdf":"https://arxiv.org/pdf/2607.16736","project":null,"code":"https://github.com/fschmid56/RealDESED","data":"https://zenodo.org/records/20056072","hfPaper":"https://huggingface.co/papers/2607.16736"},"evidence":{"snippet":"This paper presents RealDESED, a real-world domestic sound event detection (SED) benchmark comprising 5,710 audio recordings collected by 652 participants in their homes.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.16736"},"ranking":{"90d":{"score":35,"rank":201,"coverage":0.55,"confidence":"Low"}},"description":"RealDESED is a real-world domestic sound event detection benchmark with 5,710 recordings from 652 participants, 15 classes, and temporally precise annotations. Includes multi-annotator labeling and rich metadata.","whyItMatters":"Provides a realistic alternative to synthetic or web-crawled SED datasets, supporting evaluation of systems under natural domestic conditions for deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9f39c76f10dcea7cf14d317e6dd5f15c4463ed6678466c2611b2fc418a251ac1"},"motivation":"This paper presents RealDESED, a real-world domestic sound event detection (SED) benchmark comprising 5,710 audio recordings collected by 652 participants in their homes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.16736","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"The authors","organizationType":"community","sourceUrl":"https://zenodo.org/records/20056072","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_realdocbench_6da01042","familyId":"bmf_f035a1fb06db","name":"RealDocBench","oneLine":"RealDocBench evaluates field-level QA and layout understanding on real regulated documents, with 1,356 field-level questions over 581 documents and 1,500 annotated page images, scored on per-field accuracy and adjacency-aware layout metrics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.07401","pdf":"https://arxiv.org/pdf/2606.07401","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07401"},"evidence":{"snippet":"We introduce RealDocBench, a two-track benchmark built from real regulated documents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07401"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RealDocBench evaluates field-level QA and layout understanding on real regulated documents, with 1,356 field-level questions over 581 documents and 1,500 annotated page images, scored on per-field accuracy and adjacency-aware layout metrics.","whyItMatters":"It addresses the gap in document parsing evaluation by focusing on real-world regulated documents and specific field-level needs, enabling cost-aware comparisons of commercial and open-source systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a0e2f39a9d4bf85b28d3b984ddccb01bba575c8d2a795f922bae59dab2c23e51"},"motivation":"Document parsing systems are increasingly deployed in high-stakes, regulated workflows such as mortgage underwriting, financial reporting, supply-chain logistics, and clinical records.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07401","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_realistictritonbench_fb357880","familyId":"bmf_b1807ea10af3","name":"RealisticTritonBench","oneLine":"RealisticTritonBench evaluates LLM-generated Triton kernels using tasks derived from real-world pull requests in popular AI frameworks, with end-to-end integration tests.","area":"Code & Software","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12004","pdf":"https://arxiv.org/pdf/2608.12004","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12004"},"evidence":{"snippet":"To address these limitations, we introduce RealisticTritonBench, the first benchmark to derive Triton kernel generation tasks from real-world pull requests in popular AI frameworks, enabling realistic, production-like evaluation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12004"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RealisticTritonBench evaluates LLM-generated Triton kernels using tasks derived from real-world pull requests in popular AI frameworks, with end-to-end integration tests.","whyItMatters":"Existing benchmarks focus on isolated kernel translation and may have flawed evaluation scripts, while this benchmark provides realistic tasks and robust end-to-end evaluation, offering more practical insight into LLM performance for production kernel development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4245259960ec166ccdb4f1915bdc6d8dc983a5fcc26fa5b5e394d7e5c36877f0"},"motivation":"In modern AI frameworks, GPU kernels are key to overall system performance.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ASE 2026","evidence":"Accepted by ASE 2026","evidenceUrl":"https://arxiv.org/abs/2608.12004","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ASE 2026","reviewStatus":"accepted","decisionRaw":"Accepted by ASE 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.12004","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by ASE 2026","level":"author-claim"}]}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"specific"},{"id":"catalog_ce1117a4d20de18c","familyId":"catalog_family_ce1117a4d20de18c","name":"RealKIE-FCC","oneLine":"RealKIE-FCC is a key information extraction benchmark drawn from real enterprise documents (FCC filings), part of the RealKIE suite of five novel datasets for enterprise key information extraction. Models must convert documents to markdown and extract structured fields against a specified JSON schema. Nova 2 reports results on a human-verified version of the dataset.","description":"RealKIE-FCC is a key information extraction benchmark drawn from real enterprise documents (FCC filings), part of the RealKIE suite of five novel datasets for enterprise key information extraction. Models must convert documents to markdown and extract structured fields against a specified JSON schema. Nova 2 reports results on a human-verified version of the dataset.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Document Understanding","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/realkie-fcc","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ce1117a4d20de18c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/realkie-fcc"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"realkie-fcc","url":"https://llm-stats.com/benchmarks/realkie-fcc","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","document understanding","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_9c82b133d721bbd1","familyId":"catalog_family_9c82b133d721bbd1","name":"RealWorldQA","oneLine":"RealWorldQA is a benchmark designed to evaluate basic real-world spatial understanding capabilities of multimodal models. The initial release consists of over 700 anonymized images taken from vehicles and other real-world scenarios, each accompanied by a question and easily verifiable answer. Released by xAI as part of their Grok-1.5 Vision preview to test models' ability to understand natural scenes and spatial relationships in everyday visual contexts.","description":"RealWorldQA is a benchmark designed to evaluate basic real-world spatial understanding capabilities of multimodal models. The initial release consists of over 700 anonymized images taken from vehicles and other real-world scenarios, each accompanied by a question and easily verifiable answer. Released by xAI as part of their Grok-1.5 Vision preview to test models' ability to understand natural scenes and spatial relationships in everyday visual contexts.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Spatial Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9c82b133d721bbd1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/realworldqa"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/realworldqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"realWorldQa","url":"https://benchlm.ai/benchmarks/realworldqa","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"RealWorldQA","format":"Image-grounded QA","tasks":"Real-world visual question answering","successorKey":null},{"catalog":"llm-stats","sourceId":"realworldqa","url":"https://llm-stats.com/benchmarks/realworldqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","spatial reasoning","vision"],"catalogModelCount":30,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_reasonmatch-bench_7b565eab","familyId":"bmf_726b0c5520b5","name":"ReasonMatch-Bench","oneLine":"ReasonMatch-Bench evaluates wide-baseline matching and spatial reasoning in MLLMs, with benchmarks stratified by viewpoint displacement and matching granularity across indoor, outdoor, and object-centric scenarios.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03577","pdf":"https://arxiv.org/pdf/2606.03577","project":"https://aim-uofa.github.io/reasonmatch/","code":"https://github.com/aim-uofa/ReasonMatch","data":null,"hfPaper":"https://huggingface.co/papers/2606.03577"},"evidence":{"snippet":"We introduce ReasonMatch-Bench, a benchmark stratified by viewpoint displacement and matching granularity across indoor, outdoor, and object-centric scenarios, and show that current MLLMs still struggle with fine-grained wide-baseline correspondence: on a difficult 90-sample subset, human annotators achieve 84.0 F1, while the best existing baseline reaches 37.2.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":16,"hfDailySubmittedAt":null,"githubStars":19,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03577"},"ranking":{"90d":{"score":45,"rank":112,"coverage":0.7,"confidence":"Medium"}},"description":"ReasonMatch-Bench evaluates wide-baseline matching and spatial reasoning in MLLMs, with benchmarks stratified by viewpoint displacement and matching granularity across indoor, outdoor, and object-centric scenarios.","whyItMatters":"Addresses the lack of systematic evaluation for spatial reasoning in MLLMs, offering a public benchmark and reproducible training recipe to advance visual correspondence understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2361e0e31b29ed4c7cfa0919493ebca44904464faa7f0f7a2e88f51ddf52b4e2"},"motivation":"Wide-baseline matching (WBM) requires integrating geometric understanding, viewpoint changes, fine-grained perception, and occlusion reasoning, making it a challenging testbed for spatial reasoning in multimodal large language models (MLLMs) deployed in physical environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03577","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"AIM-UoFA","organizationType":"academic-lab","sourceUrl":"https://github.com/aim-uofa/ReasonMatch","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_recap_285c2bfa","familyId":"bmf_a3a6e0df57e9","name":"RECAP","oneLine":"RECAP evaluates continual-learning phenomena in prompt-based LLM adaptation under evolving constraints, using a proactive adapt-then-test protocol with constraint-level metrics for forgetting, regression, and forward transfer.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06698","pdf":"https://arxiv.org/pdf/2606.06698","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.06698"},"evidence":{"snippet":"We introduce RECAP, a benchmark that measures continual-learning phenomena (forgetting, regression, forward transfer) at the constraint level under a strictly proactive adapt-then-test protocol: prompt optimization methods receive only the constraint specification and must generalize before seeing any test data.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06698"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RECAP evaluates continual-learning phenomena in prompt-based LLM adaptation under evolving constraints, using a proactive adapt-then-test protocol with constraint-level metrics for forgetting, regression, and forward transfer.","whyItMatters":"Current benchmarks assume static constraints or reactive feedback, while real deployments often require proactive compliance. RECAP exposes performance gaps in existing prompt optimization methods under proactive adaptation, guiding development of more robust methods for evolving deployment needs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"238ac1b556da89befb2bfcf5297e6aad33a84e5afb4035faffee875f6c1f52df"},"motivation":"Production agentic systems routinely face evolving constraints and must comply from the very next interaction.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.06698","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_receipt-replay-ood_d32b05db","familyId":"bmf_172d0a1e1e4b","name":"Receipt Replay OOD","oneLine":"Receipt Replay OOD is a small out-of-domain benchmark for screen replay detection. It uses receipts, which share planar geometry, curved corners, wear-and-tear artifacts, and text patterns with identity documents, to evaluate document replay detection models under cross-domain conditions without personally identifiable information constraints.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26855","pdf":"https://arxiv.org/pdf/2605.26855","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.26855"},"evidence":{"snippet":"In this work, we introduce Receipt Replay OOD, a small out-of-domain benchmark for screen replay detection.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26855"},"ranking":{},"description":"Receipt Replay OOD is a small out-of-domain benchmark for screen replay detection. It uses receipts, which share planar geometry, curved corners, wear-and-tear artifacts, and text patterns with identity documents, to evaluate document replay detection models under cross-domain conditions without personally identifiable information constraints.","whyItMatters":"Out-of-domain robustness of screen replay detection remains underexplored, especially under realistic domain shifts. This benchmark provides a public dataset for evaluating generalization across domains, which is critical for deployment in varied presentation attack scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cd33911ed9917a71957d203e7c32ba1732ef37dc7d56cb538548cfb7b34f3d1d"},"motivation":"Public datasets such as DLC-2021, SynID, and KID34K have significantly contributed to research on presentation attack detection for identity documents, including screen replay attacks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26855","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_receiptbench_9f399a36","familyId":"bmf_268e28d589d5","name":"ReceiptBench","oneLine":"ReceiptBench evaluates multimodal large language models on visual information extraction from receipts. It includes 10k human-annotated receipts and four hierarchical subtasks: basic perception, format normalization, semantic reasoning, and structure parsing, with defined metrics and scoring.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22413","pdf":"https://arxiv.org/pdf/2605.22413","project":null,"code":"https://github.com/wwwT0ri/ReceiptBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.22413"},"evidence":{"snippet":"To bridge this gap, we introduce ReceiptBench, a large-scale, human-annotated benchmark consisting of 10k diverse receipts, organizing information extraction into four hierarchical sub-tasks: (1) Basic Perception for raw text spotting, (2) Format Normalization for strictly following standardization instructions, (3) Semantic Reasoning for inferring implicit attributes from context, and (4) Structure Parsing for handling nested line items.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.22413"},"ranking":{},"description":"ReceiptBench evaluates multimodal large language models on visual information extraction from receipts. It includes 10k human-annotated receipts and four hierarchical subtasks: basic perception, format normalization, semantic reasoning, and structure parsing, with defined metrics and scoring.","whyItMatters":"Existing VIE benchmarks lack scale, realism, and semantic granularity. ReceiptBench provides a standardized, publicly available evaluation path for comparing models on diverse receipt understanding tasks, supporting practical deployment decisions in document automation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"92f6009abe115058a572d105ce9d8bc1c0ec93cba81e4c0fff9da5794c395390"},"motivation":"Extracting structured information from visual documents (Visual Information Extraction, VIE) is a cornerstone of business automation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.22413","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"T0rl (ReceiptBench team)","organizationType":"academic-lab","sourceUrl":"https://github.com/wwwT0ri/ReceiptBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_reconstruction_90ed3fab","familyId":"bmf_d3986ad4c179","name":"Reconstruction","oneLine":"Reconstruction is a blind benchmark for recovering research ideas from pre-publication bibliographies, using a judge model to match hypotheses.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.16645","pdf":"https://arxiv.org/pdf/2608.16645","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.16645"},"evidence":{"snippet":"We introduce Reconstruction, a blind idea-recovery benchmark that withholds the seed paper and all contemporaneous or future literature, and asks models to propose hypotheses that an independent large language model judge matches against the held-out ground-truth idea.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.16645"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Reconstruction is a blind benchmark for recovering research ideas from pre-publication bibliographies, using a judge model to match hypotheses.","whyItMatters":"It tests idea recovery and anti-leakage protocols, but focuses on a specific research question rather than a general comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8bb8e1ee70111f7105a7a72c78af15c69f35906673424414b956b6664aac5074"},"motivation":"Can a language model recover the true research idea of a published paper when given only that paper's pre-publication bibliography?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.16645","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_recoveriq_46ac0f4c","familyId":"bmf_098c923d5b29","name":"recoveriq","oneLine":"The repository describes an internal evaluation harness for a revenue recovery engine, but no formal benchmark name or independent reuse protocol is declared.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Software & Cloud","Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-21","firstSeenAt":"2026-08-22","recognitionConfidence":0.4,"links":{"report":"https://github.com/aryanthemscoder/recoveriq","pdf":null,"project":"http://127.0.0.1:8000`","code":"https://github.com/aryanthemscoder/recoveriq","data":null,"hfPaper":null},"evidence":{"snippet":"recoveriq AI-assisted revenue recovery engine for failed payments with deterministic policy guardrails, simulation, and benchmark-driven evaluation.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:aryanthemscoder/recoveriq"},"ranking":{"30d":{"score":23,"rank":131,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":335,"coverage":0.55,"confidence":"Low"}},"description":"The repository describes an internal evaluation harness for a revenue recovery engine, but no formal benchmark name or independent reuse protocol is declared.","whyItMatters":"Without a stable scoring contract or public comparison path, the evaluation does not function as a benchmark for external model comparison.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"6a110b97ac5edb988b823b7ea96622a20ae7f64d90085b0409a4e28a4e2dfa58"},"motivation":"recoveriq AI-assisted revenue recovery engine for failed payments with deterministic policy guardrails, simulation, and benchmark-driven evaluation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The GitHub repository is a project README describing an internal benchmark for its own policies, lacking an independent paper or explicit formal benchmark release."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/aryanthemscoder/recoveriq","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-22T19:19:35.289219Z"},"attentionForecast":{"score":0,"confidence":"Low","horizon":"7d","reason":"Excluded or deferred due to lack of formal benchmark release."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"specific"},{"id":"catalog_1e2d96f1d3c05843","familyId":"catalog_family_1e2d96f1d3c05843","name":"RecreationBench","oneLine":"RecreationBench is Qwen's long-horizon application-recreation benchmark for hybrid agents across Ubuntu, macOS, Windows, Android, and the web.","description":"RecreationBench is Qwen's long-horizon application-recreation benchmark for hybrid agents across Ubuntu, macOS, Windows, Android, and the web.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Agents","Code","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/recreationbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1e2d96f1d3c05843"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/recreationbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"recreationbench","url":"https://llm-stats.com/benchmarks/recreationbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","agents","code","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_redact_293a3dff","familyId":"bmf_18f1255e3f73","name":"REDACT","oneLine":"REDACT is a multilingual benchmark for PII detection with 13,427 records, 51 entity types, and controlled generation axes. It allows stratified evaluation via metadata fields and includes an evaluation harness.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19881","pdf":"https://arxiv.org/pdf/2606.19881","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.19881"},"evidence":{"snippet":"We present REDACT, a systematically controlled multilingual PII benchmark with 13,427 records, 324,078 entity annotations, 51 entity types, 4,127 surface-form patterns, and 25 languages across 9 scripts.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19881"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"REDACT is a multilingual benchmark for PII detection with 13,427 records, 51 entity types, and controlled generation axes. It allows stratified evaluation via metadata fields and includes an evaluation harness.","whyItMatters":"PII detection lacks controlled benchmarks. REDACT provides systematic variation and layered evaluation to reveal failure conditions, aiding detector robustness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b0ee0792c15da051396c72af01dd24996fd7c28a84d803ee769a11951ac804f7"},"motivation":"Benchmark infrastructure for personally identifiable information (PII) detection remains limited: existing corpora cover few entity types, use ad hoc generation conditions, and do not show which surface conditions cause detector failures.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19881","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"REDACT Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.19881","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_redactionbench_13449565","familyId":"bmf_64be672a338e","name":"RedactionBench","oneLine":"RedactionBench evaluates contextual redaction of PII across 200 documents and 11 domains, with a character-level R-Score metric that treats semantically similar redactions equally.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18782","pdf":"https://arxiv.org/pdf/2606.18782","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18782"},"evidence":{"snippet":"Grounded in contextual integrity, we introduce RedactionBench, a manually annotated benchmark comprising 200 diverse documents across 11 domains, mostly seeded from real-world sources.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18782"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"RedactionBench evaluates contextual redaction of PII across 200 documents and 11 domains, with a character-level R-Score metric that treats semantically similar redactions equally.","whyItMatters":"Existing redaction benchmarks conflate extraction with privacy semantics; RedactionBench introduces contextual integrity and a metric that decouples ambiguity from precision.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bb64145d8b7ec673628d2ac2d5565c11fd1e6594793b835ac4d574d59a4818b8"},"motivation":"Large Language Models are increasingly applied to sensitive domains that require redaction of personally identifiable information (PII).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18782","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_redagentbench-executable-red-teaming-and-f_69dd1033","familyId":"bmf_8ee636d10fcd","name":"REDAgentBench","oneLine":"Evaluates LLM agent safety through executable red-teaming, adversarial case generation, and verification of harmful effects across 1,661 cases and five service surfaces.","area":"Agents & Tool Use","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.10669","pdf":"https://arxiv.org/pdf/2608.10669","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce REDAgentBench, an executable framework for autonomous red-teaming and faithful measurement.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence","named entity is a model, method, framework, or dataset used in evaluation"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10669"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLM agent safety through executable red-teaming, adversarial case generation, and verification of harmful effects across 1,661 cases and five service surfaces.","whyItMatters":"Provides an executable and measurable approach for agent safety evaluation beyond aggregate attack success rates.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"c4fa39415a4e17cf38a6b40331ca5e49368153a40cff61f2bb17ebbc93fc2c87"},"motivation":"Large language model (LLM) agents combine language-based reasoning with external tools to perform complex tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is explicitly named, defines a reproducible executable evaluation with verification, and includes a public measurement framework.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce REDAgentBench, an executable framework for autonomous red-teaming and faithful measurement"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10669","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":58,"confidence":"Medium","horizon":"7d","reason":"Agent safety and red-teaming are high-interest topics, and the benchmark's executable evaluation and measurement innovations should attract attention in the LLM safety community."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_redundancybench_b53a0300","familyId":"bmf_e290ebb79a5d","name":"RedundancyBench","oneLine":"RedundancyBench is a benchmark for detecting redundant steps in agent trajectories. It contains diverse tasks with annotated trajectories where each step is labeled for its contribution to task completion.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29893","pdf":"https://arxiv.org/pdf/2605.29893","project":"https://anonymous.4open.science/r/RedundancyBench","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29893"},"evidence":{"snippet":"To support this initiative, we introduce \\textbf{RedundancyBench}, a new benchmark that contains diverse tasks with carefully annotated trajectories, where each step is labeled according to its contribution to task completion.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29893"},"ranking":{},"description":"RedundancyBench is a benchmark for detecting redundant steps in agent trajectories. It contains diverse tasks with annotated trajectories where each step is labeled for its contribution to task completion.","whyItMatters":"LLM-based agents often execute with inefficiencies, but existing evaluations focus only on task success. RedundancyBench addresses the gap in evaluating execution efficiency.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8cad780bced7fa4ae2c5acdf2958128753af1e2c956dfb24d586bcd485740a2f"},"motivation":"LLM-based agents have demonstrated strong capabilities in solving complex tasks through multi-step reasoning and tool use.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29893","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_redvox_a5dbb799","familyId":"bmf_ecfd25a09758","name":"RedVox","oneLine":"RedVox is a multilingual safety and fairness benchmark for speech models, built on real voices. It covers unsafe and unfair stereotypical requests across five languages and evaluates models under naturalistic conditions.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26968","pdf":"https://arxiv.org/pdf/2606.26968","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.26968"},"evidence":{"snippet":"To address this gap, we introduce RedVox, a multilingual safety and fairness benchmark for audio and speech built on real voices, covering unsafe and unfair stereotypical requests across five languages (English, French, Italian, Spanish, and German).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":16,"hfDailySubmittedAt":"2026-07-01T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26968"},"ranking":{"90d":{"score":50,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"RedVox is a multilingual safety and fairness benchmark for speech models, built on real voices. It covers unsafe and unfair stereotypical requests across five languages and evaluates models under naturalistic conditions.","whyItMatters":"Speech models are deployed globally but safety evaluations are mostly English-only. RedVox provides a reusable benchmark to assess cross-lingual safety gaps and the impact of spoken input.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f54c855deb4269689c5e24c7e169df147ce34cb88d117233c7c7e6278bc0eb26"},"motivation":"Speech-capable models are increasingly deployed in real-world applications across languages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.26968","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_d38aa6a53ea7ae68","familyId":"catalog_family_d38aa6a53ea7ae68","name":"RefCOCO (avg)","oneLine":"RefCOCO-avg measures object grounding accuracy averaged across RefCOCO, RefCOCO+, and RefCOCOg benchmarks.","description":"RefCOCO-avg measures object grounding accuracy averaged across RefCOCO, RefCOCO+, and RefCOCOg benchmarks.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Spatial Reasoning","Grounding","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://github.com/lichengunc/refer","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d38aa6a53ea7ae68"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/refcocoavg"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/refcoco-avg"}],"catalogSources":[{"catalog":"benchlm","sourceId":"refcocoAvg","url":"https://benchlm.ai/benchmarks/refcocoavg","paperUrl":"https://github.com/lichengunc/refer","year":"2026","fullName":"RefCOCO average","format":"Grounded visual localization","tasks":"Referring-expression grounding","successorKey":null},{"catalog":"llm-stats","sourceId":"refcoco-avg","url":"https://llm-stats.com/benchmarks/refcoco-avg","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","spatial reasoning","grounding","vision"],"catalogModelCount":9,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_83310fc2f621b72a","familyId":"catalog_family_83310fc2f621b72a","name":"RefCOCOg","oneLine":"RefCOCOg is a referring expression comprehension benchmark that evaluates spatial grounding in images. Given a natural language expression describing an object, the model must localize the correct region, evaluated by accuracy at a 0.5 IoU threshold. It features longer, more descriptive expressions than RefCOCO and RefCOCO+.","description":"RefCOCOg is a referring expression comprehension benchmark that evaluates spatial grounding in images. Given a natural language expression describing an object, the model must localize the correct region, evaluated by accuracy at a 0.5 IoU threshold. It features longer, more descriptive expressions than RefCOCO and RefCOCO+.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Grounding","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/refcocog","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_83310fc2f621b72a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/refcocog"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"refcocog","url":"https://llm-stats.com/benchmarks/refcocog","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","grounding","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_reflexbench_c4d78abf","familyId":"bmf_0c3d9006dea6","name":"ReflexBench","oneLine":"Evaluates vision-language-action models on reaction-critical manipulation across six dynamic tasks, with configurable latency under synchronous and asynchronous inference in a simulated environment.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics","Multimodal"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.14379","pdf":"https://arxiv.org/pdf/2608.14379","project":"https://reflexvla.github.io","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.14379"},"evidence":{"snippet":"To address this gap, we present ReflexBench, a benchmark for reaction-critical manipulation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14379"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates vision-language-action models on reaction-critical manipulation across six dynamic tasks, with configurable latency under synchronous and asynchronous inference in a simulated environment.","whyItMatters":"Addresses the lack of benchmarks for dynamic interaction scenarios in robotic manipulation, providing a standardized way to assess model performance under reaction-critical conditions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"bdc4d99df748d1c2299dff99899805226400fd6c05aa0524ede64b68926d761e"},"motivation":"Vision-Language-Action (VLA) models have recently achieved promising performance in robotic manipulation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14379","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ReflexVLA Project","organizationType":"academic-lab","sourceUrl":"https://reflexvla.github.io","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_refmem-bench_c6cd0aef","familyId":"bmf_06cc5544fe08","name":"RefMem-Bench","oneLine":"RefMem-Bench is a benchmark for reflective memory in long-horizon dialogue, containing 26K QA instances across eight reflective-memory dimensions and three task formats.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01223","pdf":"https://arxiv.org/pdf/2606.01223","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01223"},"evidence":{"snippet":"To address this gap, we introduce RefMem-Bench, a benchmark for reflective memory in long-horizon dialogue.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01223"},"ranking":{},"description":"RefMem-Bench is a benchmark for reflective memory in long-horizon dialogue, containing 26K QA instances across eight reflective-memory dimensions and three task formats.","whyItMatters":"It evaluates the ability of models to synthesize fragmented multimodal cues into high-level interpretations, going beyond explicit recall.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"46107058f8122d18b9afcdd06180ee89c64834ec4a1427d54a4184e6709cd9e0"},"motivation":"Despite substantial progress in long-context modeling, existing benchmarks remain confined to factual memory for explicit recall, failing to measure the reflective memory required to synthesize fragmented, multimodal cues into high-level interpretations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01223","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_a4147f9be01ab400","familyId":"catalog_family_a4147f9be01ab400","name":"RefSpatialBench","oneLine":"RefSpatialBench evaluates spatial reference understanding and grounding.","description":"RefSpatialBench evaluates spatial reference understanding and grounding.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Spatial Reasoning","Grounding","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/refspatialbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a4147f9be01ab400"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/refspatialbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"refspatialbench","url":"https://llm-stats.com/benchmarks/refspatialbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["spatial reasoning","grounding","vision"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_refusal-benchmark_83307c24","familyId":"bmf_076cfc84d10b","name":"refusal-benchmark","oneLine":"Measures LLM refusal rates on a set of harmful prompts via OpenAI-compatible endpoints.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/audn-ai/refusal-benchmark","pdf":null,"project":"https://audn.ai/audncode","code":"https://github.com/audn-ai/refusal-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"refusal-benchmark 519+ harmful prompts to detect how abliterated AI models are # refusal-benchmark > 519+ harmful prompts to detect how abliterated AI models are.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:audn-ai/refusal-benchmark"},"ranking":{"30d":{"score":28,"rank":88,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":258,"coverage":0.55,"confidence":"Low"}},"description":"Measures LLM refusal rates on a set of harmful prompts via OpenAI-compatible endpoints.","whyItMatters":"Provides a tool for safety evaluation by quantifying model compliance with harmful requests.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"d8601bebde40d769b75ab948a189f6a5cf555935f35bf6a7a4190d9e57f02a63"},"motivation":"refusal-benchmark 519+ harmful prompts to detect how abliterated AI models are # refusal-benchmark > 519+ harmful prompts to detect how abliterated AI models are.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The repository describes a measurement tool but does not formally declare a benchmark name or provide an independent release, making its status as a benchmark unclear."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/audn-ai/refusal-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":15,"confidence":"Low","horizon":"7d","reason":"The niche focus on abliterated models and lack of formal benchmark naming likely limits initial attention."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_regiondet_ecd9208e","familyId":"bmf_57bf14a770e8","name":"RegionDet","oneLine":"RegionDet is a benchmark for region detection with eight categories, using COCO-style bounding-box annotations and evaluation protocols. It aims to extend object detection to region-level targets.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.06850","pdf":"https://arxiv.org/pdf/2608.06850","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.06850"},"evidence":{"snippet":"To address this gap, we introduce Region Detection, a task that extends conventional object detection beyond object instances, and construct RegionDet, a benchmark for region target localization.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06850"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RegionDet is a benchmark for region detection with eight categories, using COCO-style bounding-box annotations and evaluation protocols. It aims to extend object detection to region-level targets.","whyItMatters":"If published, it could support evaluation of region-level detection, but without a public release path, its standalone value is unverified.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"51a81b99f11669864ebb64e31d3955dd541862c727f48d049025ef47b5edcefb"},"motivation":"Object detection is a fundamental task in computer vision and has achieved remarkable progress on standard benchmarks by localizing discrete and well-bounded object instances.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06850","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_regretbench_a3a93900","familyId":"bmf_7ae5aa690b6f","name":"RegretBench","oneLine":"RegretBench evaluates clarification policies in multi-turn conversational LLMs, using hidden-intent tasks and a regret-based objective to measure value loss relative to a reference policy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.21143","pdf":"https://arxiv.org/pdf/2607.21143","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.21143"},"evidence":{"snippet":"We introduce RegretBench, a multi-turn benchmark that evaluates clarification as policy behavior rather than isolated question quality.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.21143"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RegretBench evaluates clarification policies in multi-turn conversational LLMs, using hidden-intent tasks and a regret-based objective to measure value loss relative to a reference policy.","whyItMatters":"It addresses the evaluation gap in conversational AI by jointly measuring intent resolution, interaction cost, and stopping decisions, offering a more comprehensive assessment of clarification behavior.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"707046653d174b7998a50f86e7495e269011a32e152dc0eefb32c930442e8746"},"motivation":"Ambiguous user requests make clarification a sequential decision problem for conversational LLM assistants: they must decide whether to ask, what to ask, when to stop, and when to answer.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.21143","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_rekey_04c2afe3","familyId":"bmf_e7eeb00b58dd","name":"REKEY","oneLine":"ReKey is a live benchmark protocol that regenerates visual keys in VQA images at evaluation time, creating fresh instances with new answers, to combat data leakage and memorization.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.20736","pdf":"https://arxiv.org/pdf/2606.20736","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.20736"},"evidence":{"snippet":"We propose ReKey, a live benchmark protocol that randomly regenerates the answer-bearing local detail, or visual key, in real images at evaluation time.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.20736"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ReKey is a live benchmark protocol that regenerates visual keys in VQA images at evaluation time, creating fresh instances with new answers, to combat data leakage and memorization.","whyItMatters":"Static benchmarks become contaminated over time; ReKey provides a contamination-resilient evaluation framework, ensuring scores reflect genuine visual ability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a8b033f353f7344a281de86ef91793c55376e835394eb1ab17eafb1682a8001d"},"motivation":"Static visual question answering (VQA) benchmarks age quickly: Once the items leak into training corpora, scores can reflect memorization rather than genuine visual ability, thus obscuring real progress.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.20736","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_rad-rule-augmented-relational-anomaly-dete_cc65accb","familyId":"bmf_04cbdb070b25","name":"Relational Anomaly Detection Benchmark","oneLine":"The Relational Anomaly Detection Benchmark evaluates anomaly detection in multi-table relational databases, spanning three settings: LANL cybersecurity events, Amazon user churn, and H&M user churn. It provides standardized tasks and evaluation protocols for relational anomaly detection methods.","area":"Vision & 3D","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":0.7,"links":{"report":"http://arxiv.org/abs/2608.23468v1","pdf":"https://arxiv.org/pdf/2608.23468v1","project":null,"code":"https://github.com/noahd15/RAD_RelationalAnomalyDetection","data":null,"hfPaper":null},"evidence":{"snippet":"To evaluate this setting, we introduce a relational anomaly detection benchmark spanning three settings: LANL cybersecurity event detection and two unexpected user-churn anomaly tasks derived from Amazon and H&M relational databases.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23468"},"ranking":{"30d":{"score":33,"rank":103,"coverage":0.55,"confidence":"Low"},"90d":{"score":29,"rank":316,"coverage":0.55,"confidence":"Low"}},"description":"The Relational Anomaly Detection Benchmark evaluates anomaly detection in multi-table relational databases, spanning three settings: LANL cybersecurity events, Amazon user churn, and H&M user churn. It provides standardized tasks and evaluation protocols for relational anomaly detection methods.","whyItMatters":"The benchmark addresses the gap in evaluating anomaly detection that preserves relational structure, which is common in real-world databases. It provides a common ground for comparing methods and advancing research in relational learning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"b28b8fd3c6c9ba76f278dc1ba7339b2ce556d3d85aa8e6ebc144f44c57133ae4"},"motivation":"Anomaly detection is often applied to data stored in relational databases, yet most existing methods require flattening multiple tables into a single feature matrix.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark includes code and data availability, a clear task definition, and standard evaluation metrics, making it reusable by other teams.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce a relational anomaly detection benchmark spanning three settings"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.23468v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":50,"confidence":"Medium","horizon":"7d","reason":"The benchmark covers security and e-commerce applications, tapping into multiple domains and potentially attracting broad interest."},"evaluationMode":"public_reusable","publishers":[{"name":"RAD Team","organizationType":"academic-lab","sourceUrl":"https://github.com/noahd15/RAD_RelationalAnomalyDetection","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_relay-bench_1dc58a62","familyId":"bmf_4a246274f907","name":"Relay-Bench","oneLine":"Relay-Bench evaluates LLMs on multi-domain reasoning chains, presenting composite problems that combine subproblems from domains like visual reasoning, coding, math, information extraction, and data analysis. It measures overall accuracy on these complex, text-only tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.18438","pdf":"https://arxiv.org/pdf/2607.18438","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.18438"},"evidence":{"snippet":"Introducing Relay-Bench, an unsaturated, holistic, text-only benchmark that measures LLMs' ability to complete an assortment of tasks from distinct domains in a single prompt.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.18438"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Relay-Bench evaluates LLMs on multi-domain reasoning chains, presenting composite problems that combine subproblems from domains like visual reasoning, coding, math, information extraction, and data analysis. It measures overall accuracy on these complex, text-only tasks.","whyItMatters":"The benchmark addresses the need for holistic evaluation of LLMs on combined reasoning across multiple domains, which is common in real-world tasks. It provides practical value in assessing models' ability to handle complex, multi-step problems, with current models scoring below 50%.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1f85cf7bede9e568ed47ccfe0ff17ebb4f859de5beedbc45cd78554f3c6b3dbe"},"motivation":"Introducing Relay-Bench, an unsaturated, holistic, text-only benchmark that measures LLMs' ability to complete an assortment of tasks from distinct domains in a single prompt.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.18438","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_remedi_11832147","familyId":"bmf_500f28f269c5","name":"REMEDI","oneLine":"REMEDI is a benchmark for machine unlearning in multi-label clinical disease inference, built on MIMIC-III, covering diverse forget sets and tasks with utility and unlearning metrics.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Safety"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07141","pdf":"https://arxiv.org/pdf/2606.07141","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07141"},"evidence":{"snippet":"To this end, we introduce REMEDI, an extensive benchmark for machine unlearning tailored to multi-label and multiclass clinical disease inference, where label correlations, longitudinal structure, and safety constraints make unlearning particularly challenging.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07141"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"REMEDI is a benchmark for machine unlearning in multi-label clinical disease inference, built on MIMIC-III, covering diverse forget sets and tasks with utility and unlearning metrics.","whyItMatters":"It provides a realistic medical-domain evaluation for machine unlearning methods, addressing the lack of benchmarks that reflect real-world patient data and multi-label scenarios, which is crucial for privacy-preserving AI in healthcare.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d8f91238afed47dda5124258f3eae926aec71f2b1d37abeec169b54bc3aa8c07"},"motivation":"Language models trained for clinical disease inference are trained on patient data, which may include sensitive and private information, and data owners may request the removal of their data from a trained model due to privacy or copyright concerns.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07141","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_remembench_10890c1e","familyId":"bmf_c308a8d726bb","name":"ReMemBench","oneLine":"ReMemBench is a benchmark with eight diverse household manipulation tasks across four categories of short-term memory, designed to evaluate memory mechanisms in visuomotor policies.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.16178","pdf":"https://arxiv.org/pdf/2606.16178","project":"https://shahrutav.github.io/short-term-memory","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.16178"},"evidence":{"snippet":"To systematically evaluate memory in visuomotor control, we introduce ReMemBench -- a benchmark of eight diverse household manipulation tasks spanning four categories of short-term memory -- designed to foster general memory mechanisms rather than siloed, task-specific solutions.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16178"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ReMemBench is a benchmark with eight diverse household manipulation tasks across four categories of short-term memory, designed to evaluate memory mechanisms in visuomotor policies.","whyItMatters":"ReMemBench addresses the lack of systematic evaluation for short-term memory in visuomotor control, providing a standardized protocol to compare memory-augmented policies and guide development for long-horizon tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d050cfd6b7237b105bd4bc8089a78db5df54dacfd5a7061f8bf6e926c5336108"},"motivation":"Many robotic tasks require short-term memory, whether it's retrieving an object that's no longer visible or turning off an appliance after a set period.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.16178","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"PRISM team","organizationType":"academic-lab","sourceUrl":"https://shahrutav.github.io/short-term-memory","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_repair-bench_816a6914","familyId":"bmf_cd82db311a00","name":"REPAIR-Bench","oneLine":"REPAIR-Bench evaluates robot error perception and recovery in human-robot interaction. It includes 214 interaction trials from 41 participants with four induced failure types, synchronized facial action units, head pose, speech transcripts, and post-interaction reports. Three tasks cover failure detection across sessions, failure-type classification, and recovery prediction.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29937","pdf":"https://arxiv.org/pdf/2606.29937","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.29937"},"evidence":{"snippet":"We present REPAIR-Bench, built on 214 interaction trials from 41 participants, the benchmark spans four induced failure types and provides synchronized facial action units, head pose, speech transcripts, and post-interaction affect and recovery reports.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29937"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"REPAIR-Bench evaluates robot error perception and recovery in human-robot interaction. It includes 214 interaction trials from 41 participants with four induced failure types, synchronized facial action units, head pose, speech transcripts, and post-interaction reports. Three tasks cover failure detection across sessions, failure-type classification, and recovery prediction.","whyItMatters":"REPAIR-Bench addresses the lack of unified benchmarks for HRI failures, enabling standardized evaluation of failure detection, classification, and recovery prediction. This supports the development of adaptive and trustworthy robot systems and provides a comparison baseline for future research.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f9dfa802988a6a09d0afd85dfbb22788fe1b83946008ad8ae198a1c97f92b50b"},"motivation":"Understanding how users perceive and respond to robot failures is essential for building robust and trustworthy robot systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.29937","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_repbench_7ff35873","familyId":"bmf_b37ff0b1a335","name":"RepBench","oneLine":"RepBench compiles 353 public benchmarks into 46,149 probe texts spanning 94 capabilities, with a taxonomy of 182 clusters in 13 families. It evaluates representation readout methods across 12 models under cross-benchmark transfer, providing a reusable closed-loop pipeline for capability-aligned probing.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-30","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.28008","pdf":"https://arxiv.org/pdf/2607.28008","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.28008"},"evidence":{"snippet":"We present RepBench, a benchmark-grounded data layer for capability-aligned representation probing.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.28008"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RepBench compiles 353 public benchmarks into 46,149 probe texts spanning 94 capabilities, with a taxonomy of 182 clusters in 13 families. It evaluates representation readout methods across 12 models under cross-benchmark transfer, providing a reusable closed-loop pipeline for capability-aligned probing.","whyItMatters":"RepBench addresses the lack of comparable and reproducible evaluation for representation engineering by grounding probes in multiple public benchmarks, reducing single-source bias and enabling meaningful comparison of readout methods and aggregation criteria across models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d8bab187c774e445ecd97f0923d704197d7c5ce7afbffd994fc236376409ea4a"},"motivation":"Representation engineering reads and steers capability directions in large language models, yet methods are typically evaluated on paper-specific synthetic data.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.28008","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_dbd95a95416ed37b","familyId":"catalog_family_dbd95a95416ed37b","name":"Repo Env","oneLine":"Repo Env evaluates an agent's ability to set up, configure, and run real repositories, including dependency resolution and environment management.","description":"Repo Env evaluates an agent's ability to set up, configure, and run real repositories, including dependency resolution and environment management.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/repo-env","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dbd95a95416ed37b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/repo-env"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"repo-env","url":"https://llm-stats.com/benchmarks/repo-env","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","coding"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_repo-context-ambiguity_46658557","familyId":"bmf_9459abdb9856","name":"repo-context-ambiguity","oneLine":"Measures whether code generation models violate implicit repository conventions, using automatic AST oracles for constraint and security violations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-23","firstSeenAt":"2026-08-24","recognitionConfidence":0.4,"links":{"report":"https://github.com/Parfait18/repo-context-ambiguity","pdf":null,"project":null,"code":"https://github.com/Parfait18/repo-context-ambiguity","data":null,"hfPaper":null},"evidence":{"snippet":"A benchmark with automatic AST oracles.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:parfait18/repo-context-ambiguity"},"ranking":{"30d":{"score":23,"rank":114,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":318,"coverage":0.55,"confidence":"Low"}},"description":"Measures whether code generation models violate implicit repository conventions, using automatic AST oracles for constraint and security violations.","whyItMatters":"Addresses a gap between prompt ambiguity and real-world codebase constraints, but lacks independent evidence of formal release.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"1d375917ce6ec82bee0f4318f3b7b7356b66d5e696d116bb22947f25b1d1a2ba"},"motivation":"repo-context-ambiguity Do code generation models respect constraints that live in the codebase rather than in the prompt?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"New GitHub repository with no paper or independent verification; benchmark identity and stability not sufficiently established for publication."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/Parfait18/repo-context-ambiguity","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-24T07:04:32.475961Z"},"attentionForecast":{"score":10,"confidence":"Low","horizon":"7d","reason":"No external visibility signals in the input, and a single repository discovery is unlikely to generate significant early attention."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_d5676b4212a9258a","familyId":"catalog_family_d5676b4212a9258a","name":"RepoBench","oneLine":"RepoBench is a benchmark for evaluating repository-level code auto-completion systems through three interconnected tasks: RepoBench-R (retrieval of relevant code snippets across files), RepoBench-C (code completion with cross-file and in-file context), and RepoBench-P (pipeline combining retrieval and prediction). Supports Python and Java programming languages and addresses the gap in evaluating real-world, multi-file programming scenarios by providing a more complete comparison of performance in auto-completion systems.","description":"RepoBench is a benchmark for evaluating repository-level code auto-completion systems through three interconnected tasks: RepoBench-R (retrieval of relevant code snippets across files), RepoBench-C (code completion with cross-file and in-file context), and RepoBench-P (pipeline combining retrieval and prediction). Supports Python and Java programming languages and addresses the gap in evaluating real-world, multi-file programming scenarios by providing a more complete comparison of performance in auto-completion systems.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/repobench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d5676b4212a9258a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/repobench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"repobench","url":"https://llm-stats.com/benchmarks/repobench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_repomirage_322a2daa","familyId":"bmf_defa1b39bb7c","name":"RepoMirage","oneLine":"RepoMirage is a two-stage evaluation suite built on SWE-Bench Verified that applies semantics-preserving repository-level perturbations and extended tasks to probe repository context reasoning in code agents.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26177","pdf":"https://arxiv.org/pdf/2605.26177","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.26177"},"evidence":{"snippet":"To investigate this question, we introduce RepoMirage, a two-stage evaluation suite built on SWE-Bench Verified that adopts perturbation as a diagnostic tool to increase the demand for context reasoning by transforming how the repository is exposed.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26177"},"ranking":{},"description":"RepoMirage is a two-stage evaluation suite built on SWE-Bench Verified that applies semantics-preserving repository-level perturbations and extended tasks to probe repository context reasoning in code agents.","whyItMatters":"It aims to isolate repository context reasoning from end-to-end issue resolution performance, revealing a gap that could inform structure-aware agent design.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cad8c803504208bfc383b6e69168d3ddfababa4d8939051acf7b1c39d5ced261"},"motivation":"Code agents are currently having skillful performance on repository-level software engineering benchmarks, but it remains unclear whether success on end-to-end tasks such as issue resolution truly reflects repository context reasoning, the ability to identify the task-relevant information across multiple files and reason over the relations among them.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26177","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_4de85080d39833b1","familyId":"catalog_family_4de85080d39833b1","name":"RepoQA","oneLine":"RepoQA is a benchmark for evaluating long-context code understanding capabilities of Large Language Models through the Searching Needle Function (SNF) task, where LLMs must locate specific functions in code repositories using natural language descriptions. The benchmark contains 500 code search tasks spanning 50 repositories across 5 modern programming languages (Python, Java, TypeScript, C++, and Rust), tested on 26 general and code-specific LLMs to assess their ability to comprehend and navigate code repositories.","description":"RepoQA is a benchmark for evaluating long-context code understanding capabilities of Large Language Models through the Searching Needle Function (SNF) task, where LLMs must locate specific functions in code repositories using natural language descriptions. The benchmark contains 500 code search tasks spanning 50 repositories across 5 modern programming languages (Python, Java, TypeScript, C++, and Rust), tested on 26 general and code-specific LLMs to assess their ability to comprehend and navigate code repositories.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/repoqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4de85080d39833b1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/repoqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"repoqa","url":"https://llm-stats.com/benchmarks/repoqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering","Long Context & Memory"],"domainScope":"general"},{"id":"bm_reporeasoner_6d64b1f4","familyId":"bmf_edc74b50c095","name":"RepoReasoner","oneLine":"Evaluates long-context LLMs on repository-level code reasoning through Output Prediction and Call Chain Prediction tasks, using dynamic tracing and I/O rewriting to reduce memorization.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Reasoning"],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25996","pdf":"https://arxiv.org/pdf/2607.25996","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.25996"},"evidence":{"snippet":"We introduce RepoReasoner, a benchmark for evaluating repository-level code reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25996"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates long-context LLMs on repository-level code reasoning through Output Prediction and Call Chain Prediction tasks, using dynamic tracing and I/O rewriting to reduce memorization.","whyItMatters":"Assesses cross-file reasoning capabilities that are critical for real-world software engineering, identifying limitations beyond function-level benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"9e134949cc2337612f71aabfaf0e36c56bd9e5aacd353607585221c1faa5fbd8"},"motivation":"Recent large language models (LLMs) have shown strong performance on software engineering tasks, yet most existing benchmarks evaluate code reasoning at the function level, where all relevant information is localized.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25996","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"RepoReasoner Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.25996","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering","Long Context & Memory"],"domainScope":"general"},{"id":"bm_reref-3d_34628627","familyId":"bmf_dc0caa547026","name":"ReRef-3D","oneLine":"ReRef-3D benchmarks language-guided 3D scene rearrangement with 33,826 instructions across 998 scenes, evaluating placement validity and relation satisfaction.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.16011","pdf":"https://arxiv.org/pdf/2608.16011","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.16011"},"evidence":{"snippet":"We introduce ReRef-3D, a benchmark for language-guided placement in 3D scenes.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.16011"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ReRef-3D benchmarks language-guided 3D scene rearrangement with 33,826 instructions across 998 scenes, evaluating placement validity and relation satisfaction.","whyItMatters":"It targets spatial reasoning in embodied AI and provides metrics for relation satisfaction and physical validity in rearrangement tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"74e540dbdf23f203f9aad76c7ad4b20439432416f33fe25962c5be48b0b68579"},"motivation":"We introduce ReRef-3D, a benchmark for language-guided placement in 3D scenes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.16011","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_rescast-100k_da1d15f3","familyId":"bmf_4fbc0d9dc8e9","name":"RESCAST-100K","oneLine":"RESCAST-100K evaluates cross-domain residential load and indoor temperature forecasting. It provides ~100,000 EnergyPlus-simulated U.S. homes with 15-minute time series for total load, HVAC load, and indoor temperature, plus weather, setpoints, and static covariates. It includes configurable domain axes and integrates five real-world datasets for sim-to-real evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02852","pdf":"https://arxiv.org/pdf/2606.02852","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02852"},"evidence":{"snippet":"We introduce RESCAST-100K, a large-scale residential forecasting benchmark for studying cross-domain generalization.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02852"},"ranking":{},"description":"RESCAST-100K evaluates cross-domain residential load and indoor temperature forecasting. It provides ~100,000 EnergyPlus-simulated U.S. homes with 15-minute time series for total load, HVAC load, and indoor temperature, plus weather, setpoints, and static covariates. It includes configurable domain axes and integrates five real-world datasets for sim-to-real evaluation.","whyItMatters":"Existing residential forecasting datasets are narrow and lack structured cross-domain evaluation. RESCAST-100K enables systematic assessment of transfer learning and domain adaptation under controlled shifts, supporting better generalization in home energy management and grid-scale applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8e337ce9dffe970f3cefac4f15113cbbf5cc0b410b2f0bd60a079ba229d41361"},"motivation":"Accurate short-term forecasting of residential energy load and indoor temperature is essential for home energy management systems, grid-level demand response, and community energy efficiency efforts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02852","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_rescuebench_f2263257","familyId":"bmf_6b6323c23009","name":"RescueBench","oneLine":"RescueBench evaluates embodied search-and-rescue agents in simulated photo-realistic environments. It comprises a four-stage pipeline: multimodal exploration, target rescue, memory-guided return, and final handoff. Five difficulty levels vary environmental complexity, clue ambiguity, and spatial hierarchy. Automatic episode generation and annotation support scalable evaluation. A unified benchmark framework and runner scripts provide standardized scoring.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01848","pdf":"https://arxiv.org/pdf/2606.01848","project":null,"code":"https://github.com/wukui-muc/RescueBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.01848"},"evidence":{"snippet":"We introduce RescueBench, a photo-realistic diagnostic benchmark that instantiates SAR as a four-stage pipeline: multimodal exploration, target rescue, memory-guided return, and final handoff.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01848"},"ranking":{},"description":"RescueBench evaluates embodied search-and-rescue agents in simulated photo-realistic environments. It comprises a four-stage pipeline: multimodal exploration, target rescue, memory-guided return, and final handoff. Five difficulty levels vary environmental complexity, clue ambiguity, and spatial hierarchy. Automatic episode generation and annotation support scalable evaluation. A unified benchmark framework and runner scripts provide standardized scoring.","whyItMatters":"Search-and-rescue benchmarks typically test capabilities in isolation. RescueBench addresses the gap of composite workflows where failures may compound across stages. It provides stage-level diagnostics to identify bottlenecks (e.g., exploration, memory) separately from end-to-end performance, helping practitioners target improvements in embodied agents for realistic rescue tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"64a76691b682d95054288a0a3cf96cc73246b329fb3ee052d76fc4361bf07ea3"},"motivation":"Search-and-rescue (SAR) requires embodied agents to explore unfamiliar environments under multimodal uncertainty, perform multi-stage interactions, and retrieve spatial memory over long horizons.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01848","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_researchclawbench_4d57a097","familyId":"bmf_4fbc99503d1c","name":"ResearchClawBench","oneLine":"ResearchClawBench evaluates autonomous scientific research agents across 40 tasks from 10 domains, each grounded in a published paper with provided literature and raw data. It uses expert-curated multimodal rubrics to score target-paper-level re-discovery.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07591","pdf":"https://arxiv.org/pdf/2606.07591","project":null,"code":"https://github.com/InternScience/ResearchClawBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.07591"},"evidence":{"snippet":"We present ResearchClawBench, a benchmark for evaluating autonomous scientific research across 40 tasks from 10 scientific domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":102,"hfDailySubmittedAt":null,"githubStars":252,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07591"},"ranking":{},"description":"ResearchClawBench evaluates autonomous scientific research agents across 40 tasks from 10 domains, each grounded in a published paper with provided literature and raw data. It uses expert-curated multimodal rubrics to score target-paper-level re-discovery.","whyItMatters":"Autonomous research agents claim to accelerate science, but their end-to-end capability is unverified. ResearchClawBench provides a standardized evaluation frontier to measure progress toward reliable research re-discovery.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"76aa8b49483bc5675f84dd79a9cdb2fb4dea680995d1ef2abf931cd7dd935e20"},"motivation":"AI coding agents are increasingly used for scientific work, but their end-to-end autonomous research capability remains difficult to verify.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07591","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"InternScience","organizationType":"community","sourceUrl":"https://github.com/InternScience/ResearchClawBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"researchClawBench","url":"https://benchlm.ai/benchmarks/researchclawbench","paperUrl":"https://arxiv.org/abs/2606.07591","year":"2026","fullName":"ResearchClawBench","format":"End-to-end autonomous research evaluation with RADS scoring","tasks":"40 tasks across 10 scientific domains","successorKey":null},{"catalog":"llm-stats","sourceId":"researchclawbench","url":"https://llm-stats.com/benchmarks/researchclawbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","research","agents","tool calling"],"catalogModelCount":1,"catalogStarCount":0},{"id":"bm_researchqa_6b10c7bd","familyId":"bmf_27dd5af619ba","name":"ResearchQA","oneLine":"ResearchQA is a benchmark of 6,211 single-paper question-answer pairs from 494 open-access papers across eight domains, designed for citation-grounded evaluation with multiple valid supporting passages and grounded refusal.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.11074","pdf":"https://arxiv.org/pdf/2607.11074","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.11074"},"evidence":{"snippet":"We introduce ResearchQA, a benchmark of 6,211 single-paper question-answer pairs from 494 open-access papers spanning eight domains and four question types: lookup, comprehension, multi-hop, and adversarial.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.11074"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ResearchQA is a benchmark of 6,211 single-paper question-answer pairs from 494 open-access papers across eight domains, designed for citation-grounded evaluation with multiple valid supporting passages and grounded refusal.","whyItMatters":"Existing evaluation methods often fail to detect whether answers are supported by verifiable citations. ResearchQA provides a standardized way to assess citation accuracy and groundedness, separating systems more clearly than LLM-evaluator scores.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"294ca04530c165a5a5deb556ef6549c260be9dc19623174031b01b9c05ba1bd9"},"motivation":"Large language models are increasingly used to assist scientific reading, but existing evaluation methods often fail to detect whether answers are supported by verifiable citations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.11074","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_retailbench_1e87a5e2","familyId":"bmf_9a9797d125ac","name":"RetailBench","oneLine":"RetailBench is a simulation benchmark evaluating tool-using LLM agents in single-store supermarket operations over a 180-day horizon. It covers pricing, replenishment, supplier selection, assortment, inventory aging, customer feedback, external events, and cash-flow constraints, with a privileged oracle policy for comparison.","area":"Language & Knowledge","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Logistics"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15862","pdf":"https://arxiv.org/pdf/2606.15862","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.15862"},"evidence":{"snippet":"We introduce RetailBench, a data-grounded simulation benchmark for evaluating tool-using LLM agents in single-store supermarket operation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15862"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RetailBench is a simulation benchmark evaluating tool-using LLM agents in single-store supermarket operations over a 180-day horizon. It covers pricing, replenishment, supplier selection, assortment, inventory aging, customer feedback, external events, and cash-flow constraints, with a privileged oracle policy for comparison.","whyItMatters":"RetailBench addresses the gap in evaluating LLM agents on long-horizon, economically grounded decision-making, where short-horizon tasks dominate existing benchmarks. It provides a controlled testbed for assessing coherent decision-making and reliability in dynamic environments.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4cb9393ccb61d88fc5a48fcfeeac409e38663318a036a3fe9a0c2277a5b06b7d"},"motivation":"Large language model (LLM) agents have made rapid progress on short-horizon, well-scoped tasks, yet their ability to sustain coherent decisions in dynamic long-horizon environments remains uncertain.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15862","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_revengebench_23ac692d","familyId":"bmf_928072ade2b4","name":"RevengeBench","oneLine":"RevengeBench is a benchmark for recovering code-space policies from behavioral traces. It includes 75 LLM-generated policies across five game environments, where a learner designs behavioral probes and submits executable hypotheses, evaluated using continuous action-distance metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.26094","pdf":"https://arxiv.org/pdf/2606.26094","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.26094"},"evidence":{"snippet":"We introduce RevengeBench, a benchmark of 75 LLM generated, Elo-calibrated policies across five game environments, drawn from CodeClash tournament trajectories.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26094"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"RevengeBench is a benchmark for recovering code-space policies from behavioral traces. It includes 75 LLM-generated policies across five game environments, where a learner designs behavioral probes and submits executable hypotheses, evaluated using continuous action-distance metrics.","whyItMatters":"Addresses the inverse problem of inferring hidden decision programs from observations, relevant to opponent modeling and policy interpretability. It provides a tractable testbed for studying how controlled experiments improve code-space recovery.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"42b247ad81d5bd6e9df43d160eb22e5216533c64ab9133092e848e1d86399656"},"motivation":"For most of scientific history, researchers studying behavior could only infer hidden mechanisms from outward actions: an inverse problem that becomes more tractable when observation is augmented by targeted intervention.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.26094","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_70090a3bc9b98ec7","familyId":"catalog_family_70090a3bc9b98ec7","name":"ReverseEngBench","oneLine":"A contamination-resistant agent benchmark for reverse engineering real-world binaries.","description":"A contamination-resistant agent benchmark for reverse engineering real-world binaries.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/reverse_eng","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_70090a3bc9b98ec7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsreverseengbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsReverseEngBench","url":"https://benchlm.ai/benchmarks/valsreverseengbench","paperUrl":"https://www.vals.ai/benchmarks/reverse_eng","year":"2026","fullName":"Vals ReverseEngBench","format":"Fully solved rate and capability score","tasks":"Real-world binary reverse-engineering tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_revico_34e02af8","familyId":"bmf_3ee719669886","name":"ReViCo","oneLine":"Vision Language Models (VLMs) have shown great success in general visual tasks, yet they still struggle to deeply understand text within images.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.27154","pdf":"https://arxiv.org/pdf/2608.27154","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.27154"},"evidence":{"snippet":"In this paper, we introduce ReViCo (Real Visual Correction), a benchmark designed to evaluate VLM text understanding through a novel task of visual text error correction.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27154"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"motivation":"Vision Language Models (VLMs) have shown great success in general visual tasks, yet they still struggle to deeply understand text within images.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.27154","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_5606bc30b998befe","familyId":"catalog_family_5606bc30b998befe","name":"RiemannBench (no tools)","oneLine":"Research-level mathematics problems with unique programmatically verified closed-form answers.","description":"Research-level mathematics problems with unique programmatically verified closed-form answers.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5606bc30b998befe"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/riemannbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"riemannBench","url":"https://benchlm.ai/benchmarks/riemannbench","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"RiemannBench without tools","format":"Accuracy without tools","tasks":"25 private research-level mathematics problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_c6997e7d4c6f7040","familyId":"catalog_family_c6997e7d4c6f7040","name":"RiemannBench (tools)","oneLine":"Research-level mathematics problems with unique programmatically verified closed-form answers and tool access.","description":"Research-level mathematics problems with unique programmatically verified closed-form answers and tool access.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c6997e7d4c6f7040"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/riemannbenchwithtools"}],"catalogSources":[{"catalog":"benchlm","sourceId":"riemannBenchWithTools","url":"https://benchlm.ai/benchmarks/riemannbenchwithtools","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"RiemannBench with tools","format":"Accuracy with tools","tasks":"25 private research-level mathematics problems","successorKey":null}],"catalogCategories":["math"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_rift-bench_5db6cac9","familyId":"bmf_140e541b846e","name":"RIFT-Bench","oneLine":"RIFT-Bench is described as a methodology for dynamic red-teaming of agentic AI systems, using a graph representation to enable unified evaluations across diverse agentic architectures. It operates in two phases: Discovery and Scanning, deploying adaptive adversarial attacks and producing evaluation reports.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.23927","pdf":"https://arxiv.org/pdf/2606.23927","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.23927"},"evidence":{"snippet":"To address this gap, we introduce RIFT-Bench, a graph representation-driven methodology for dynamic red-teaming that enables unified evaluations across diverse agentic architectures.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.23927"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RIFT-Bench is described as a methodology for dynamic red-teaming of agentic AI systems, using a graph representation to enable unified evaluations across diverse agentic architectures. It operates in two phases: Discovery and Scanning, deploying adaptive adversarial attacks and producing evaluation reports.","whyItMatters":"As agentic AI systems become more autonomous, they introduce new attack surfaces beyond traditional LLM vulnerabilities. RIFT-Bench aims to provide a scalable foundation for security evaluation across heterogeneous agentic architectures, which could aid in comparing the robustness of different systems and evaluating mitigation strategies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e40cc81229cc15d976980edd192b68f9c210a046ae7072b970707b7ca3672362"},"motivation":"Agentic AI systems powered by large language models (LLMs) are rapidly evolving into autonomous decision-making systems, exposing attack vectors beyond those of traditional LLM vulnerabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.23927","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_rigidbench_f95c26d5","familyId":"bmf_c9cf6af36be4","name":"RigidBench","oneLine":"RigidBench evaluates rigid-body physics in video generation models using a simulator-grounded benchmark with 100 examples and five tasks. It provides ten measurements covering motion, geometry, identity, background stability, and appearance, with per-frame data.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15555","pdf":"https://arxiv.org/pdf/2608.15555","project":"https://doi.org/10.5281/zenodo.21649156","code":"https://github.com/swarnim-j/RigidBench","data":null,"hfPaper":"https://huggingface.co/papers/2608.15555"},"evidence":{"snippet":"We introduce RigidBench, a simulator-grounded benchmark that compares a generated continuation with a reference rollout from the same initial frame and motion description.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15555"},"ranking":{"30d":{"score":28,"rank":94,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":264,"coverage":0.55,"confidence":"Low"}},"description":"RigidBench evaluates rigid-body physics in video generation models using a simulator-grounded benchmark with 100 examples and five tasks. It provides ten measurements covering motion, geometry, identity, background stability, and appearance, with per-frame data.","whyItMatters":"Video generation metrics often mix independent errors. RigidBench separates these aspects and shows rankings depend on what is measured, providing a more detailed evaluation for physics fidelity.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"687a65c5c85f54877092b3b4c0c0049c51bfdb67e9bad140c78ca8974a31e4f6"},"motivation":"Video models are increasingly used to predict what happens next in a scene, yet the metrics commonly used to compare their outputs say little about whether the predicted objects move correctly.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15555","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"RigidBench contributors","organizationType":"academic-lab","sourceUrl":"https://github.com/swarnim-j/RigidBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_rigorbench_122c9d4f","familyId":"bmf_41aa52836b4e","name":"RigorBench","oneLine":"RigorBench evaluates autonomous AI coding agents on engineering process discipline across five pillars: Planning Fidelity, Verification Coverage, Recovery Efficiency, Abstention Quality, and Atomic Transition Integrity. It includes 30 tasks in five categories and a composite RigorScore metric.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22678","pdf":"https://arxiv.org/pdf/2606.22678","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.22678"},"evidence":{"snippet":"We introduce RigorBench, the first benchmark designed to measure process discipline in AI coding agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22678"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"RigorBench evaluates autonomous AI coding agents on engineering process discipline across five pillars: Planning Fidelity, Verification Coverage, Recovery Efficiency, Abstention Quality, and Atomic Transition Integrity. It includes 30 tasks in five categories and a composite RigorScore metric.","whyItMatters":"Existing agent benchmarks focus on outcome correctness, ignoring process quality. RigorBench fills this gap by measuring how agents plan, verify, and recover, providing a more comprehensive assessment for reliable deployment in real-world software engineering.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b307b8177198c0f9921172b8e9e4fc4c0170e57c3b449c8940f317cd19278d17"},"motivation":"Agentic coding harnesses - such as Agent-Skills, Superpowers, and Agent-Rigor - are increasingly deployed to augment underlying LLMs for real-world software engineering tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22678","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_rng-bench_1fb2eb04","familyId":"bmf_4e8d6f247af1","name":"RNG-Bench","oneLine":"RNG-Bench evaluates multimodal LLMs in controllable non-Markov games, requiring reconstruction of past observations and acting on them, with two games: Matching Pairs and 3D Maze.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19338","pdf":"https://arxiv.org/pdf/2606.19338","project":null,"code":"https://github.com/InternLM/RNGBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.19338"},"evidence":{"snippet":"We introduce RNG-Bench (Reconstructive Non-Markov Games), a benchmark suite designed to isolate a base model's ability to reconstruct past observations and act on them during multi-step interaction.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":51,"hfDailySubmittedAt":"2026-06-18T00:00:00.000Z","githubStars":41,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19338"},"ranking":{"90d":{"score":54,"rank":44,"coverage":0.7,"confidence":"Medium"}},"description":"RNG-Bench evaluates multimodal LLMs in controllable non-Markov games, requiring reconstruction of past observations and acting on them, with two games: Matching Pairs and 3D Maze.","whyItMatters":"Addresses the gap in evaluating models' abilities to remember and act on hidden state, which is critical for real-world deployments where observations are partial.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"47e264b08d3d68b5084f6a5e801e641cd2e92b8b8cc6c9ac7434569b9803aade"},"motivation":"Deploying multimodal foundation models as closed-loop policies increasingly requires conditioning actions on observations that are no longer visible.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19338","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"InternLM","organizationType":"benchmark-organization","sourceUrl":"https://github.com/InternLM/RNGBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_robodojo_1fa2182f","familyId":"bmf_1b4dac6f946e","name":"RoboDojo","oneLine":"RoboDojo is a unified sim-and-real benchmark for evaluating generalist robot manipulation policies, comprising 42 simulation tasks and 18 real-world tasks across three robot embodiments, covering five capability dimensions: generalization, memory, precision, long-horizon execution, and open-vocabulary instruction following.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.04434","pdf":"https://arxiv.org/pdf/2607.04434","project":"http://robodojo-benchmark.com/","code":"https://github.com/RoboDojo-Benchmark/RoboDojo","data":"https://huggingface.co/datasets/RoboDojo-Benchmark/RoboDojo","hfPaper":"https://huggingface.co/papers/2607.04434"},"evidence":{"snippet":"We introduce RoboDojo, a unified sim-and-real benchmark for comprehensive evaluation of generalist robot manipulation policies.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":16,"hfDailySubmittedAt":"2026-07-09T00:00:00.000Z","githubStars":445,"githubScope":"benchmark_repo","hfDatasetDownloads":175690,"hfDatasetLikes":8},"source":{"type":"arxiv","id":"2607.04434"},"ranking":{"90d":{"score":83,"rank":1,"coverage":1.0,"confidence":"High","datasetDownloadRank":1,"datasetRankPopulation":66}},"description":"RoboDojo is a unified sim-and-real benchmark for evaluating generalist robot manipulation policies, comprising 42 simulation tasks and 18 real-world tasks across three robot embodiments, covering five capability dimensions: generalization, memory, precision, long-horizon execution, and open-vocabulary instruction following.","whyItMatters":"Existing benchmarks often limit scope to narrow tasks or single settings. RoboDojo provides a comprehensive, reproducible evaluation across simulation and real world, enabling systematic comparison of policies through a public leaderboard and cloud-based real-world evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e24539efe71fb44b9e46c2e09dea791b0d155e38e01b2a86d5552b438dbcf5b5"},"motivation":"Generalist robot manipulation policies have advanced rapidly, yet existing benchmarks remain limited in systematically evaluating their capabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"source-reviewed","reviewedAt":"2026-08-30","sources":["https://arxiv.org/abs/2607.04434","https://github.com/RoboDojo-Benchmark/RoboDojo","https://huggingface.co/datasets/RoboDojo-Benchmark/RoboDojo"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.04434","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"RoboDojo Benchmark Team","organizationType":"academic-lab","sourceUrl":"http://robodojo-benchmark.com/","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_robomme-interference_7b518e1a","familyId":"bmf_f0a1facb41ae","name":"RoboMME-Interference","oneLine":"RoboMME-Interference evaluates robot long-context memory under cross-session interference. The benchmark builds on RoboMME, constructing session histories per query episode with relevant demonstration plus controlled unrelated sessions, and measures task success for memory-augmented vision-language-action models.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22338","pdf":"https://arxiv.org/pdf/2606.22338","project":"https://robotmemorybench.com","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.22338"},"evidence":{"snippet":"To measure how current robot memory systems perform on longer sessions with more distractions, we introduce RoboMME-Interference, a cross-session benchmark built on RoboMME.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22338"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"RoboMME-Interference evaluates robot long-context memory under cross-session interference. The benchmark builds on RoboMME, constructing session histories per query episode with relevant demonstration plus controlled unrelated sessions, and measures task success for memory-augmented vision-language-action models.","whyItMatters":"Existing robot memory benchmarks ignore realistic multi-session interference. RoboMME-Interference quantifies how memory decays with unrelated sessions and whether retrieval mechanisms restore robustness, providing practical decision value for long-deployed robots.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b9d6394cb1ee7992ed3105c080ada8703802ee95f9e1baebb3eaabc55a46ca8e"},"motivation":"Robots deployed in realistic settings will accumulate experience across many sessions and tasks over their deployment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22338","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"robotmemorybench.com","organizationType":"academic-lab","sourceUrl":"https://robotmemorybench.com","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_roboprocessbench_629af272","familyId":"bmf_51d8097c561d","name":"RoboProcessBench","oneLine":"RoboProcessBench evaluates vision-language models on process-aware understanding in robotic manipulation, with 12 diagnostic question families covering static monitoring and dynamic reasoning over execution traces. The benchmark includes 58k QA pairs across 260 tasks.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics","Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13040","pdf":"https://arxiv.org/pdf/2606.13040","project":"https://processbench-2026.github.io/RoboProcessBench-Web/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.13040"},"evidence":{"snippet":"To address this gap, we present RoboProcessBench, a benchmark for process-aware understanding in vision-language robotic manipulation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13040"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"RoboProcessBench evaluates vision-language models on process-aware understanding in robotic manipulation, with 12 diagnostic question families covering static monitoring and dynamic reasoning over execution traces. The benchmark includes 58k QA pairs across 260 tasks.","whyItMatters":"Existing evaluations largely ignore fine-grained process understanding, which is crucial for VLMs used as critics or failure detectors. This benchmark provides a structured way to measure progress in this capability and supports post-training via a dedicated SFT split.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9967320f876bc6201dd5f2694cb322a41863de83122a184d7a74dcd26496a1e0"},"motivation":"Vision-language models (VLMs) are increasingly explored as visual critics, reward generators, and failure detectors in robotic manipulation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13040","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"RoboProcessBench Project","organizationType":"community","sourceUrl":"https://processbench-2026.github.io/RoboProcessBench-Web/","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_robosemanticbench_9bca3113","familyId":"bmf_85bece8521c2","name":"RoboSemanticBench","oneLine":"RoboSemanticBench (RSB) is an embodied benchmark evaluating whether vision-language-action models use instruction semantics to select and grasp the correct physical target among candidate blocks in response to math or general-knowledge questions. It includes six suites with four- or ten-choice variants, procedural arithmetic, GSM8K-style, and MMLU-style questions, with diagnostic metrics separating task success from grasp success.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02277","pdf":"https://arxiv.org/pdf/2606.02277","project":null,"code":"https://github.com/ZGC-EmbodyAI/RoboSemanticBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.02277"},"evidence":{"snippet":"We introduce RoboSemanticBench (RSB), an embodied benchmark for diagnosing semantic grounding in action prediction: whether post-trained VLA models can use complex instruction semantics to select and manipulate the correct physical target.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":7,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02277"},"ranking":{},"description":"RoboSemanticBench (RSB) is an embodied benchmark evaluating whether vision-language-action models use instruction semantics to select and grasp the correct physical target among candidate blocks in response to math or general-knowledge questions. It includes six suites with four- or ten-choice variants, procedural arithmetic, GSM8K-style, and MMLU-style questions, with diagnostic metrics separating task success from grasp success.","whyItMatters":"RSB addresses the evaluation gap of measuring whether VLA models actually ground instruction semantics in action prediction, separating low-level manipulation from semantic selection. It provides a controlled, repeatable protocol with held-out questions, useful for diagnosing model capabilities and guiding improvements in embodied semantic understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"587e7e6271eb53d0d9a42c28e6a342f95702b75866ea73cb01aba2d2069c3b3b"},"motivation":"Vision-language-action (VLA) models are built on the premise that semantic understanding from pretrained language or vision-language backbones should guide robot action prediction.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02277","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ZGC-EmbodyAI","organizationType":"academic-lab","sourceUrl":"https://github.com/ZGC-EmbodyAI/RoboSemanticBench","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_c837d8ede0c41b52","familyId":"catalog_family_c837d8ede0c41b52","name":"RoboSpatialHome","oneLine":"RoboSpatialHome evaluates spatial understanding for robotic home navigation and manipulation.","description":"RoboSpatialHome evaluates spatial understanding for robotic home navigation and manipulation.","area":"Mathematical Reasoning","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":[],"capabilities":[],"topics":["Robotics","Spatial Reasoning","Embodied","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/robospatialhome","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c837d8ede0c41b52"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/robospatialhome"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"robospatialhome","url":"https://llm-stats.com/benchmarks/robospatialhome","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["robotics","spatial reasoning","embodied","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"bm_robotrustbench_454946c5","familyId":"bmf_2309dd08a194","name":"RoboTrustBench","oneLine":"RoboTrustBench evaluates the trustworthiness of video world models for robotic manipulation across four scenarios: Normal, Constraint-Sensitive, Counterfactual, and Adversarial. It contains 1,207 expert-validated instruction-image pairs from DROID episodes and a six-dimensional evaluation protocol with 13 fine-grained criteria.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01600","pdf":"https://arxiv.org/pdf/2606.01600","project":"https://huiqiongli.github.io/RoboTrustBench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01600"},"evidence":{"snippet":"We introduce RoboTrustBench, a benchmark for evaluating the trustworthiness of video world models under four scenarios: Normal, Constraint-Sensitive, Counterfactual, and Adversarial.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01600"},"ranking":{},"description":"RoboTrustBench evaluates the trustworthiness of video world models for robotic manipulation across four scenarios: Normal, Constraint-Sensitive, Counterfactual, and Adversarial. It contains 1,207 expert-validated instruction-image pairs from DROID episodes and a six-dimensional evaluation protocol with 13 fine-grained criteria.","whyItMatters":"Existing benchmarks for video world models largely overlook trustworthiness aspects such as constraint reasoning, counterfactual grounding, and safety. RoboTrustBench provides a structured evaluation to assess these capabilities, offering practical guidance for selecting and improving models for safe robotic manipulation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"15548aa1d2295b999b236d057b81f446c1a22b8293446670ef77579a87a37f08"},"motivation":"Video world models are increasingly used in robotic manipulation, yet existing benchmarks mostly evaluate them under valid, feasible, and safe instructions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01600","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"RoboTrustBench Team","organizationType":"academic-lab","sourceUrl":"https://huiqiongli.github.io/RoboTrustBench/","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_robotvalues_5349608d","familyId":"bmf_2f76fccff529","name":"RobotValues","oneLine":"RobotValues evaluates household robot planners in 10K value-conflict scenarios with realistic images, testing value preferences and action selection under conflicting human values.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03312","pdf":"https://arxiv.org/pdf/2606.03312","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03312"},"evidence":{"snippet":"We introduce RobotValues, a benchmark to evaluate household robot planners in 10K value-conflict scenarios.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":26,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03312"},"ranking":{"90d":{"score":51,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"RobotValues evaluates household robot planners in 10K value-conflict scenarios with realistic images, testing value preferences and action selection under conflicting human values.","whyItMatters":"No public artifacts or reuse path are provided, and the benchmark lacks a clear scoring contract for independent runs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2f55a891c5398618b076eb837b0978c8e42a2805acaf974ac79b81ac8aac0467"},"motivation":"While household robots are often evaluated based on task completion, everyday domestic environments involve value-conflicting situations in which robots are expected to choose actions that prioritize other values than task success, such as human autonomy, efficiency, or social appropriateness.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03312","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_robovista_4b4e3ac7","familyId":"bmf_3df0c5f42047","name":"RoboVista","oneLine":"RoboVista evaluates vision-language models on robot question answering with 474 visual question answering instances spanning 39 task types across agricultural, industrial, domestic, surgical, and autonomous driving domains.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.04610","pdf":"https://arxiv.org/pdf/2607.04610","project":"https://berkeleyautomation.github.io/robovista/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.04610"},"evidence":{"snippet":"We propose Robot Question Answering (RQA), a modular evaluation framework and RoboVista, a benchmark curated from real robotic systems, research papers, and expert annotations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.04610"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RoboVista evaluates vision-language models on robot question answering with 474 visual question answering instances spanning 39 task types across agricultural, industrial, domestic, surgical, and autonomous driving domains.","whyItMatters":"Robot applications require modular reasoning across diverse embodiments. This benchmark isolates decision components and shows correlation with real-world task execution, aiding VLM selection.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"309dd1beb54804291eca7401b61f4e0b37428c9ffacb78a60113abaf55e93255"},"motivation":"Diverse applications for robotics, such as industry and agriculture, require robots to operate across various embodiments, changing visual conditions, and complex planning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"RSS 2026","evidence":"Accepted to RSS 2026. Project website: https://berkeleyautomation.github.io/robovista/","evidenceUrl":"https://arxiv.org/abs/2607.04610","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"RSS 2026","reviewStatus":"accepted","decisionRaw":"Accepted to RSS 2026. Project website: https://berkeleyautomation.github.io/robovista/","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.04610","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to RSS 2026. Project website: https://berkeleyautomation.github.io/robovista/","level":"author-claim"}]}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_robowits_1755c069","familyId":"bmf_dfdca2555adf","name":"RoboWits","oneLine":"RoboWits is a bi-manual robotic benchmark designed to evaluate cognitive reasoning, creative tool use, and robustness to unexpected conditions. It includes 30 seed tasks and 208 mutated tasks with graded difficulty across geometry, material, and assembly-based reasoning.","area":"Agents & Tool Use","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Reasoning","Tool use","Robustness"],"topics":["Agents","Reasoning"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30326","pdf":"https://arxiv.org/pdf/2605.30326","project":"https://umass-embodied-agi.github.io/RoboWits","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30326"},"evidence":{"snippet":"We introduce RoboWits, a bi-manual robotic benchmark designed to systematically evaluate cognitive reasoning, creative tool use, and robustness to unexpected conditions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30326"},"ranking":{},"description":"RoboWits is a bi-manual robotic benchmark designed to evaluate cognitive reasoning, creative tool use, and robustness to unexpected conditions. It includes 30 seed tasks and 208 mutated tasks with graded difficulty across geometry, material, and assembly-based reasoning.","whyItMatters":"Current robotic benchmarks focus on skill-level execution, not the cognitive reasoning needed for real-world adaptation. RoboWits aims to evaluate reasoning-centric capabilities under unexpected challenges.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"62b701a819189a501668cd0b71bb780789cd7c6770e17bc63f9b46ecfd64230b"},"motivation":"The ability to reason, adapt, and creatively solve problems under unexpected challenges is essential for robots operating in real-world environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30326","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents","Tool Calling"],"domainScope":"specific"},{"id":"catalog_94b164de4437e42a","familyId":"catalog_family_94b164de4437e42a","name":"Robust IF","oneLine":"Robust IF evaluates instruction-following robustness on diverse, hard prompts, measuring whether a model reliably adheres to constraints across challenging single-turn and multi-turn scenarios.","description":"Robust IF evaluates instruction-following robustness on diverse, hard prompts, measuring whether a model reliably adheres to constraints across challenging single-turn and multi-turn scenarios.","area":"Instruction Following","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Instruction Following","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/robust-if","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_94b164de4437e42a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/robust-if"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"robust-if","url":"https://llm-stats.com/benchmarks/robust-if","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["instruction following","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Instruction Following & Structured Output"],"domainScope":"general"},{"id":"bm_robustmad_0c7403de","familyId":"bmf_191732065a96","name":"RobustMAD","oneLine":"Evaluates robustness of multimodal small language models for industrial anomaly detection across open-ended queries and visual degradations, using multiple-choice accuracy and LLM-judged open-ended responses.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Robustness"],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.16243","pdf":"https://arxiv.org/pdf/2607.16243","project":"https://openreview.net/forum?id=skrA9UYNIZ","code":"https://github.com/en-research/RobustMAD","data":null,"hfPaper":"https://huggingface.co/papers/2607.16243"},"evidence":{"snippet":"To address this gap, we develop RobustMAD, the first deployment-motivated benchmark, designed to comprehensively evaluate model robustness through diverse open-ended queries spanning object understanding, anomaly detection, unanswerable problems, and visual quality degradations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.16243"},"ranking":{"90d":{"score":31,"rank":224,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates robustness of multimodal small language models for industrial anomaly detection across open-ended queries and visual degradations, using multiple-choice accuracy and LLM-judged open-ended responses.","whyItMatters":"Assesses deployability of compact models in real-world industrial conditions, identifying failure modes like fragile grounding and hallucination on ill-posed queries.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6ea21137d1a73d54364f7629546ff74194bead95c73b6f9b122de37c977e15d3"},"motivation":"Multimodal industrial anomaly inspection assistants are a critical component of next-generation smart factories, enabling interactive vision-language-based querying.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"publication in Transactions on Machine Learning Research (TMLR), 2026 at https://openreview","evidence":"Accepted for publication in Transactions on Machine Learning Research (TMLR), 2026 at https://openreview.net/forum?id=skrA9UYNIZ","evidenceUrl":"https://arxiv.org/abs/2607.16243","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"publication in Transactions on Machine Learning Research (TMLR), 2026 at https://openreview","reviewStatus":"accepted","decisionRaw":"Accepted for publication in Transactions on Machine Learning Research (TMLR), 2026 at https://openreview.net/forum?id=skrA9UYNIZ","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.16243","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted for publication in Transactions on Machine Learning Research (TMLR), 2026 at https://openreview.net/forum?id=skrA9UYNIZ","level":"author-claim"}]}],"publishers":[{"name":"EN Research","organizationType":"academic-lab","sourceUrl":"https://github.com/en-research/RobustMAD","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_rolecde_7c0f211a","familyId":"bmf_ce6bd9dfe8c8","name":"RoleCDE","oneLine":"RoleCDE evaluates role-playing agents under structured conflicts between role-specific values and alignment-oriented constraints. It comprises approximately 8,000 role profiles and 240,000 dilemma instances across three difficulty levels and eight role categories, with scoring via LLM-as-a-judge.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01552","pdf":"https://arxiv.org/pdf/2606.01552","project":null,"code":"https://github.com/rabbitrose/RoleCDE","data":null,"hfPaper":"https://huggingface.co/papers/2606.01552"},"evidence":{"snippet":"To address this gap, we introduce RoleCDE, the first benchmark designed to evaluate RPAs under structured conflicts between role-specific values and alignment-oriented constraints.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01552"},"ranking":{},"description":"RoleCDE evaluates role-playing agents under structured conflicts between role-specific values and alignment-oriented constraints. It comprises approximately 8,000 role profiles and 240,000 dilemma instances across three difficulty levels and eight role categories, with scoring via LLM-as-a-judge.","whyItMatters":"Existing benchmarks focus on surface fidelity and lack coverage of decision-making under role-alignment value conflicts. RoleCDE provides a systematic evaluation of how agents resolve such conflicts, revealing systematic biases and offering a tool for improving alignment and role consistency.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c20da1b38ee79974202a55eca82e8b9607810ea76939e9cab77f375de94bd6df"},"motivation":"Role-playing agents(RPAs) are widely used to steer large language models(LLMs) toward role-consistent behavior, yet existing benchmarks mainly evaluate surface-level fidelity and offer limited insight into decision making under role-alignment value conflicts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01552","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"RoleCDE Team","organizationType":"academic-lab","sourceUrl":"https://github.com/rabbitrose/RoleCDE","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_roman-numeral-harmony-benchmark_f8f54472","familyId":"bmf_4a091afad75c","name":"Roman Numeral Harmony Benchmark","oneLine":"Evaluates Roman-numeral analysis, cadence classification, and key identification across four synthetic notation representations with gold deterministic labels.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/4esv/roman-numeral-harmony-benchmark","pdf":null,"project":"https://huggingface.co/datasets/4esv/roman-numeral-harmony-benchmark","code":"https://github.com/4esv/roman-numeral-harmony-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"roman-numeral-harmony-benchmark A benchmark for functional harmony: Roman-numeral analysis, cadence classification and key identification, in four representations.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:4esv/roman-numeral-harmony-benchmark"},"ranking":{"30d":{"score":23,"rank":118,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":322,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates Roman-numeral analysis, cadence classification, and key identification across four synthetic notation representations with gold deterministic labels.","whyItMatters":"It isolates music-theory reasoning from performance and enables controlled studies of how representation affects model competence.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"ad7f27cc83c34c35a36cdf37719b4e637758df527b7546febd970faa39be3281"},"motivation":"roman-numeral-harmony-benchmark A benchmark for functional harmony: Roman-numeral analysis, cadence classification and key identification, in four representations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/4esv/roman-numeral-harmony-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":35,"confidence":"Low","horizon":"7d","reason":"A specialized symbolic-music benchmark is likely to attract a focused niche audience rather than broad early attention."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_rotoutbench_a7d65303","familyId":"bmf_67067fe3674d","name":"RotOutBench","oneLine":"Paired diagnostic benchmark for rotated-outcome prediction in vision-language models, spanning open visual cases and controlled text-image rotations, with accuracy metrics for direct reading and prediction.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.07641","pdf":"https://arxiv.org/pdf/2606.07641","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07641"},"evidence":{"snippet":"To isolate this gap, we introduce RotOutBench, a paired diagnostic benchmark spanning open visual cases and controlled text-image rotations.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07641"},"ranking":{},"description":"Paired diagnostic benchmark for rotated-outcome prediction in vision-language models, spanning open visual cases and controlled text-image rotations, with accuracy metrics for direct reading and prediction.","whyItMatters":"Evaluates a specific cognitive ability in VLMs, but lacks a standalone public comparison path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"621eed638574e244809bd38bf77a7430150c1d406be542f2ddae745a35773eed"},"motivation":"Can vision-language models predict what a 180{\\deg} rotation would reveal from the original image alone?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07641","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_routebench_6fe659bf","familyId":"bmf_16ecd82fb9ee","name":"ROUTEBENCH","oneLine":"ROUTEBENCH evaluates whether transformers can learn latent algorithm routing across four solver families (ridge, lasso, Huber, kNN) in a controlled diagnostic setting.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Robustness"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.24471","pdf":"https://arxiv.org/pdf/2607.24471","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.24471"},"evidence":{"snippet":"We introduce ROUTEBENCH, a diagnostic benchmark whose regimes differentially favor global shrinkage, sparsity, robustness, and locality, operationalized by ridge-like, lasso-like, Huber-like, and kNN-like family representatives.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24471"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ROUTEBENCH evaluates whether transformers can learn latent algorithm routing across four solver families (ridge, lasso, Huber, kNN) in a controlled diagnostic setting.","whyItMatters":"Provides controlled evidence on internal routing in transformers, but lacks a standalone public comparison path and is primarily a research probe.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f67a2cb06183b1d8085e2266d7c81b1f360e6a534ec56ce0b6f83ad1302ea8f2"},"motivation":"A central question in the in-context learning literature is whether transformers can organize episode-level adaptation around different inductive-bias families.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"COLM 2026","evidence":"Accepted by COLM 2026","evidenceUrl":"https://arxiv.org/abs/2607.24471","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"COLM 2026","reviewStatus":"accepted","decisionRaw":"Accepted by COLM 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.24471","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by COLM 2026","level":"author-claim"}]}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_rq-bench_0360bc38","familyId":"bmf_c9112d4b22f3","name":"RQ-Bench","oneLine":"RQ-Bench evaluates novelty of research questions generated by LLMs against author-anchored reference questions from recent arXiv papers, using standalone and comparative LLM judging as well as human expert evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.DL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-10","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.12071","pdf":"https://arxiv.org/pdf/2606.12071","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.12071"},"evidence":{"snippet":"We introduce RQ-Bench, a benchmark built from recent arXiv papers.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":"2026-06-10T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.12071"},"ranking":{"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"RQ-Bench evaluates novelty of research questions generated by LLMs against author-anchored reference questions from recent arXiv papers, using standalone and comparative LLM judging as well as human expert evaluation.","whyItMatters":"Addresses the reliability of LLM-based novelty assessment for scientific ideation, indicating that LLM judges may produce a 'novelty mirage' compared to human experts. Useful for researchers evaluating automated scientific review or generation systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8e03f9b7c9de33ce9f875f99b46d0b781095724daff49d7d062b1b2e5c6ba26a"},"motivation":"LLMs are increasingly used to generate and judge scientific ideas.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.12071","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_rrs-10k_71143844","familyId":"bmf_76690795f4ea","name":"RRS-10K","oneLine":"RRS-10K is a benchmark for rare remote sensing image interpretation containing 10,738 military-related images with multiple format question-answer pairs, organized into three capability dimensions and 20 leaf tasks covering perception, reasoning, and robustness.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.24810","pdf":"https://arxiv.org/pdf/2607.24810","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.24810"},"evidence":{"snippet":"To address this gap, we present RRS-10K, a benchmark for rare remote sensing image interpretation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24810"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RRS-10K is a benchmark for rare remote sensing image interpretation containing 10,738 military-related images with multiple format question-answer pairs, organized into three capability dimensions and 20 leaf tasks covering perception, reasoning, and robustness.","whyItMatters":"Current remote sensing benchmarks are dominated by common scenes, limiting understanding of VLM performance on rare, long-tail scenarios. RRS-10K provides a standardized evaluation for this gap, enabling systematic analysis of failure modes and guiding development of more reliable remote sensing VLMs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9a49ac6bb876401f97e9665c5c08748ea21e5812646d387fed08ec59b23ec2b4"},"motivation":"Vision-language models (VLMs) have achieved strong performance on general remote sensing tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24810","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_rs-rie-bench_f4bb954a","familyId":"bmf_71850eabb7fa","name":"RS-RIE-Bench","oneLine":"RS-RIE-Bench evaluates reasoning-guided remote sensing image editing across temporal, causal, and spatial reasoning tasks, with metrics for region plausibility, preservation, and quality consistency.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20197","pdf":"https://arxiv.org/pdf/2607.20197","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20197"},"evidence":{"snippet":"To fill this gap, we introduce RS-RIE-Bench, the first benchmark for reasoning-guided remote sensing image editing.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20197"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RS-RIE-Bench evaluates reasoning-guided remote sensing image editing across temporal, causal, and spatial reasoning tasks, with metrics for region plausibility, preservation, and quality consistency.","whyItMatters":"Current image editing benchmarks focus on natural images, leaving a gap for remote sensing domains that require geographic reasoning and sensor consistency. A dedicated benchmark supports progress in specialized editing models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b14d8d6d408ab61824df1d42d792769a04517dc92294389f72b5c235f2d03768"},"motivation":"Remote sensing image editing aims to modify remote sensing images according to natural language instructions while preserving geographic rules and sensor observation characteristics.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20197","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_f2d3e1af3fd43d61","familyId":"catalog_family_f2d3e1af3fd43d61","name":"RSI Index","oneLine":"The RSI (Recursive Self-Improvement) Index is OpenAI's aggregate metric across a bundle of internal AI-research evaluations, including debugging research systems, optimizing kernels and training recipes, and improving other models, measuring progress toward recursive self-improvement.","description":"The RSI (Recursive Self-Improvement) Index is OpenAI's aggregate metric across a bundle of internal AI-research evaluations, including debugging research systems, optimizing kernels and training recipes, and improving other models, measuring progress toward recursive self-improvement.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Agents","Code","Systems"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/rsi-index","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f2d3e1af3fd43d61"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/rsi-index"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"rsi-index","url":"https://llm-stats.com/benchmarks/rsi-index","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","code","systems"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_rsibench-data_1ace1cc0","familyId":"bmf_e3268eaf0064","name":"RSIBench-Data","oneLine":"RSIBench-Data evaluates LLM agents as data-centric researchers, where agents iteratively revise training-data strategies for a fixed target model on six benchmarks, with real training and evaluation runs.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Self-Evolution"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25886","pdf":"https://arxiv.org/pdf/2607.25886","project":null,"code":"https://github.com/evolvent-ai/RSIBench-Data","data":null,"hfPaper":"https://huggingface.co/papers/2607.25886"},"evidence":{"snippet":"We introduce RSIBench-Data, a controlled benchmark of LLM agents as data-centric researchers with a fixed post-training stack.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":139,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25886"},"ranking":{"90d":{"score":60,"rank":20,"coverage":0.55,"confidence":"Low"}},"description":"RSIBench-Data evaluates LLM agents as data-centric researchers, where agents iteratively revise training-data strategies for a fixed target model on six benchmarks, with real training and evaluation runs.","whyItMatters":"It isolates research capability from engineering, showing that current agents can improve from feedback but inconsistently. This provides an auditable testbed for capabilities needed in recursive self-improvement.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8433842326a170ba278916ef9854aff1083d43680966b7a32be6b16c16d1ee6c"},"motivation":"Recursive self-improvement requires turning evidence of model failures into better models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25886","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"evolvent-ai","organizationType":"company-research-lab","sourceUrl":"https://github.com/evolvent-ai/RSIBench-Data","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_rtcbench_9cd327fe","familyId":"bmf_dd843208c5ec","name":"RTCbench","oneLine":"Evaluates LLM-generated controllers on simulated closed-loop control tasks using metrics like CVaR@10%, safety gating, and replayable traces.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/aiast1/rtcbench","pdf":null,"project":null,"code":"https://github.com/aiast1/rtcbench","data":null,"hfPaper":null},"evidence":{"snippet":"rtcbench A benchmark for LLMs that commission real closed-loop control systems.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:aiast1/rtcbench"},"ranking":{"30d":{"score":23,"rank":142,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":346,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates LLM-generated controllers on simulated closed-loop control tasks using metrics like CVaR@10%, safety gating, and replayable traces.","whyItMatters":"Provides a rigorous evaluation for control system commissioning with safety gates, bad-tail ranking, and falsifiable scoring.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"11701b6a97ec2712745f1631a270abf6a65a324c9cc20dd7460b0538d8b37588"},"motivation":"rtcbench A benchmark for LLMs that commission real closed-loop control systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/aiast1/rtcbench","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"attentionForecast":{"score":25,"confidence":"Low","horizon":"7d","reason":"Specialized control systems niche with modest reach, but existing leaderboard and reproducibility may attract some attention."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_rtl-bench_4e1de99d","familyId":"bmf_d6fea90599a9","name":"RTL-Bench","oneLine":"RTL-BenchLS evaluates LLMs on RTL design generation and reasoning, containing over 10,000 formally verified Verilog designs. Tasks include specification-to-RTL generation, round-trip reasoning, masked-content reasoning, and repository-issue reasoning. All tasks are verified via formal equivalence checking without manual testbenches.","area":"Language & Knowledge","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Semiconductors"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08976","pdf":"https://arxiv.org/pdf/2606.08976","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.08976"},"evidence":{"snippet":"We introduce RTL-BenchLS, a large-scale benchmark addressing both limitations above.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08976"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RTL-BenchLS evaluates LLMs on RTL design generation and reasoning, containing over 10,000 formally verified Verilog designs. Tasks include specification-to-RTL generation, round-trip reasoning, masked-content reasoning, and repository-issue reasoning. All tasks are verified via formal equivalence checking without manual testbenches.","whyItMatters":"Existing RTL benchmarks are small and saturate with frontier models. RTL-BenchLS provides a large-scale, challenging benchmark with self-supervised tasks, enabling tracking of progress on complex hardware design reasoning and generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"177e66c8502318178b5e2467d723a9ae68b8c370faee281d179fb09c165b7248"},"motivation":"LLM-based RTL generation and reasoning is a promising direction for hardware design automation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08976","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_rubench_84878dc8","familyId":"bmf_8cbfda054603","name":"RuBench","oneLine":"Evaluates coding agents on 25 repository-level tasks in Russian, mined from recent fix commits across five open-source projects. Graded by upstream regression tests with withheld oracles. Multiple rounds document model change and contamination audits.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.06411","pdf":"https://arxiv.org/pdf/2607.06411","project":null,"code":"https://github.com/eugeneshilow/rubench","data":null,"hfPaper":"https://huggingface.co/papers/2607.06411"},"evidence":{"snippet":"We introduce RuBench 1.0, a benchmark of 25 tasks mined from recent fix commits in five live open-source repositories (aiohttp, aiogram, Laravel, NestJS, Fastify), each specified natively in Russian -- written from scratch, not translated -- and judged by the upstream maintainer's regression tests, which we withhold from release.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06411"},"ranking":{"90d":{"score":25,"rank":298,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates coding agents on 25 repository-level tasks in Russian, mined from recent fix commits across five open-source projects. Graded by upstream regression tests with withheld oracles. Multiple rounds document model change and contamination audits.","whyItMatters":"Provides a benchmark with natively authored non-English specifications, addressing a gap in multilingual agent evaluation. Includes rigorous auditing and honest scores, which matter for reliability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a9be3960d9a6af537aba4350cba5c298e1b5bd51dd87ec7ba0a52f8444c6a29e"},"motivation":"Developers increasingly delegate real maintenance work to product-grade coding agents, and many state tasks in their native language, in the style of a customer request rather than a curated English issue.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06411","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Evgeny Shilov","organizationType":"community","sourceUrl":"https://github.com/eugeneshilow/rubench","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_rule-compliant-visual-spatial-planning-for_2793101a","familyId":"bmf_1b6b5724735c","name":"RuleMaze","oneLine":"Benchmark for rule-compliant visual spatial planning in multimodal LLMs, requiring maze navigation under natural-language rules with automated rule generation and validation.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-21","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.20237","pdf":"https://arxiv.org/pdf/2608.20237","project":null,"code":"https://github.com/oceanflowlab/RuleMaze","data":null,"hfPaper":null},"evidence":{"snippet":"To address this gap, we introduce RuleMaze, a controllable benchmark in which MLLMs must navigate mazes while obeying natural-language rules of varying complexity.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.20237"},"ranking":{"30d":{"score":54,"rank":53,"coverage":0.55,"confidence":"Low"},"90d":{"score":43,"rank":236,"coverage":0.55,"confidence":"Low"}},"description":"Benchmark for rule-compliant visual spatial planning in multimodal LLMs, requiring maze navigation under natural-language rules with automated rule generation and validation.","whyItMatters":"Fills a gap in evaluating MLLMs on joint visual perception, rule interpretation, and constrained action planning, with scalable rule construction and a public leaderboard.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"9e4db56f22d9ace1af1ad77b037cf7dd1825cd943b50bc4b9777ab926d772925"},"motivation":"Multimodal large language models (MLLMs) combine linguistic reasoning with visual perception, yet their ability to perform visual spatial planning under explicit or previously unseen rule constraints remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The paper introduces a benchmark with code, dataset, and evaluation pipeline, supporting ongoing model comparison.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce RuleMaze, a controllable benchmark in which MLLMs must navigate mazes while obeying natural-language rules"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.20237","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-21T04:30:09.569171Z"},"attentionForecast":{"score":70,"confidence":"Medium","horizon":"7d","reason":"Multimodal LLM planning is a hot topic and the benchmark provides code, dataset, and a project page, likely generating broad interest."},"evaluationMode":"score_submission","publishers":[{"name":"OceanFlowLab","organizationType":"academic-lab","sourceUrl":"https://github.com/oceanflowlab/RuleMaze","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"lib_ruler","familyId":"family_ruler","name":"RULER","oneLine":"Established benchmark family · Long Context.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2404.06654","pdf":null,"project":"https://github.com/NVIDIA/RULER","code":"https://github.com/NVIDIA/RULER","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_ruler"},"ranking":{},"recordType":"family","aliases":[],"sourceAttribution":[{"role":"official-repository","url":"https://github.com/NVIDIA/RULER"}],"adoptionRefs":["google-gemini25"],"modelReportReferences":[{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"ruler","url":"https://llm-stats.com/benchmarks/ruler","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":4,"catalogStarCount":0},{"id":"catalog_3dbe558adb85db84","familyId":"catalog_family_3dbe558adb85db84","name":"RULER 1000K","oneLine":"RULER 1000K evaluates the official 13-task RULER v1 suite at a 1048576-token (1M) context budget.","description":"RULER 1000K evaluates the official 13-task RULER v1 suite at a 1048576-token (1M) context budget.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ruler-1000k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3dbe558adb85db84"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ruler-1000k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ruler-1000k","url":"https://llm-stats.com/benchmarks/ruler-1000k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_af1a13b9070aa1d9","familyId":"catalog_family_af1a13b9070aa1d9","name":"RULER 128k","oneLine":"RULER 128k evaluates the official 13-task RULER v1 suite at a 131072-token context budget.","description":"RULER 128k evaluates the official 13-task RULER v1 suite at a 131072-token context budget.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ruler-128k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_af1a13b9070aa1d9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ruler-128k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ruler-128k","url":"https://llm-stats.com/benchmarks/ruler-128k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_99591d2231898d4e","familyId":"catalog_family_99591d2231898d4e","name":"RULER 2048K","oneLine":"RULER 2048K evaluates the official 13-task RULER v1 suite at a 2097152-token (2M) context budget.","description":"RULER 2048K evaluates the official 13-task RULER v1 suite at a 2097152-token (2M) context budget.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ruler-2048k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_99591d2231898d4e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ruler-2048k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ruler-2048k","url":"https://llm-stats.com/benchmarks/ruler-2048k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_cab5dec64ae20822","familyId":"catalog_family_cab5dec64ae20822","name":"RULER 512K","oneLine":"RULER 512K evaluates the official 13-task RULER v1 suite at a 524288-token context budget.","description":"RULER 512K evaluates the official 13-task RULER v1 suite at a 524288-token context budget.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ruler-512k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_cab5dec64ae20822"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ruler-512k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ruler-512k","url":"https://llm-stats.com/benchmarks/ruler-512k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_d772247e3b0e0811","familyId":"catalog_family_d772247e3b0e0811","name":"RULER 64k","oneLine":"RULER 64k evaluates the official 13-task RULER v1 suite at a 65536-token context budget.","description":"RULER 64k evaluates the official 13-task RULER v1 suite at a 65536-token context budget.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/ruler-64k","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d772247e3b0e0811"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/ruler-64k"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"ruler-64k","url":"https://llm-stats.com/benchmarks/ruler-64k","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_ruleshift-bench_b86e99e6","familyId":"bmf_a4b2fc88cb45","name":"RuleShift-Bench","oneLine":"RuleShift-Bench is a benchmark for evaluating concept drift under evolving concept definitions, spanning multiple data types and revision types. It assesses the ability of learning systems to adapt to rule-induced concept shifts with high accuracy while minimizing reprocessing.","area":"Language & Knowledge","applicationDomains":["Cybersecurity","Finance & Economics"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity","Financial Services"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-26","recognitionConfidence":0.65,"links":{"report":"https://arxiv.org/abs/2608.23893","pdf":"https://arxiv.org/pdf/2608.23893","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We also introduce RuleShift-Bench, spanning financial, demographic, cybersecurity, and graph-structured data with threshold, predicate, logical, relational, recurring, and mixed concept revisions.","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23893"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"RuleShift-Bench is a benchmark for evaluating concept drift under evolving concept definitions, spanning multiple data types and revision types. It assesses the ability of learning systems to adapt to rule-induced concept shifts with high accuracy while minimizing reprocessing.","whyItMatters":"The benchmark addresses the challenge of adapting learning systems to explicit concept revisions, which is common in real-world deployed systems. It provides a standardized evaluation to compare methods for incremental learning and data maintenance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"8561521e6bf7c183e317ac03c589db12807ca62a2ad495f34d78af269afc3eca"},"motivation":"Learning systems deployed over long periods must adapt not only to statistical changes in incoming data, but also to revisions of the definitions that generate their prediction targets.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The paper introduces a named benchmark with specified domains and revision types, and indicates release, meeting the benchmark criteria.","canonicalNameSource":"abstract","canonicalNameEvidence":"We also introduce RuleShift-Bench, spanning financial, demographic, cybersecurity, and graph-structured data with threshold, predicate, logical, relational, recurring, and mixed concept revisions."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.23893","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":45,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses a relevant problem in machine learning and data management, with broad applicability across domains."},"evaluationMode":"public_reusable","publishers":[{"name":"RuleShift-Bench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.23893","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"bm_ruleweaver-benchmarking-rule-centered-scen_2379b2dc","familyId":"bmf_5a6c19eca015","name":"RuleWeaver","oneLine":"Evaluates rule-centered scenario reasoning through corpus-derived IF-THEN rules, composed into QA instances with rubric-based answer quality, rule recall, and rule precision scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-27","firstSeenAt":"2026-08-30","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.26832","pdf":"https://arxiv.org/pdf/2608.26832","project":null,"code":"https://github.com/SharkSpicy-NLP/RuleWeaver","data":null,"hfPaper":null},"evidence":{"snippet":"To address these gaps, this paper introduces RuleWeaver, a benchmark construction framework for evaluating rule-centered scenario reasoning.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL","named entity is a model, method, framework, or dataset used in evaluation"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":null,"hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26832"},"ranking":{},"description":"Evaluates rule-centered scenario reasoning through corpus-derived IF-THEN rules, composed into QA instances with rubric-based answer quality, rule recall, and rule precision scoring.","whyItMatters":"Provides process-level evaluation beyond final-answer correctness, exposing where models fail to apply complex rules in domain scenarios.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T07:23:36.749451Z","inputHash":"5cf98e41bd1413778c84681f04bff9ae4e392e5c3341d3f6dbbd619fa723b5fd"},"motivation":"Large language models (LLMs) are increasingly applied to specialized domains, where effective use of domain expertise often requires reasoning over complex rules in concrete scenarios.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T07:23:36.749451Z","model":"deepseek-v4-pro","decisionReason":"Named benchmark with released code and dataset, formal scoring rubrics, and a public GitHub path for reproduction.","canonicalNameSource":"abstract","canonicalNameEvidence":"this paper introduces RuleWeaver, a benchmark construction framework for evaluating rule-centered scenario reasoning"},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 Findings","evidence":"Accepted by EMNLP 2026 Findings","evidenceUrl":"https://arxiv.org/abs/2608.26832","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-30T06:59:55.242506Z"},"venueAttempts":[{"venueName":"EMNLP 2026 Findings","reviewStatus":"accepted","decisionRaw":"Accepted by EMNLP 2026 Findings","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.26832","observedAt":"2026-08-30T06:59:55.242506Z","rawValue":"Accepted by EMNLP 2026 Findings","level":"author-claim"}]}],"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"Accepted at EMNLP 2026 Findings and available code are likely to generate interest among LLM evaluation researchers."},"evaluationMode":"public_reusable","publishers":[{"name":"NLP Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/SharkSpicy-NLP/RuleWeaver","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_beyond-factual-knowledge-benchmarking-and-_bf5f8184","familyId":"bmf_ace1cc273de1","name":"RuleWorld","oneLine":"Evaluates step-level procedural rule reasoning with single-rule, parallel multi-rule, and multi-hop scenarios over large rule pools.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Factuality"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":0.7,"links":{"report":"http://arxiv.org/abs/2608.22753v1","pdf":"https://arxiv.org/pdf/2608.22753v1","project":null,"code":"https://github.com/SharkSpicy-NLP/Beyond-Factual-Knowledge","data":null,"hfPaper":null},"evidence":{"snippet":"To evaluate this capability, we introduce RuleWorld, a large-scale benchmark that reformulates rules as globally reusable abstract units rather than instance-specific facts.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22753"},"ranking":{"today":{"score":50,"rank":7,"coverage":0.95,"confidence":"High"},"30d":{"score":50,"rank":69,"coverage":0.85,"confidence":"High"},"90d":{"score":45,"rank":229,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates step-level procedural rule reasoning with single-rule, parallel multi-rule, and multi-hop scenarios over large rule pools.","whyItMatters":"Tests whether models can apply externally provided rules at scale, a capability needed for reliable procedural reasoning in real-world tasks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"e9f000a22a56b1fbf9ece448403765db9dde6b1989c9278f53b621af035dd44d"},"motivation":"Large language models (LLMs) excel at text understanding and generation, yet still struggle to reliably understand and apply externally provided procedural rules at scale.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with dataset released on Hugging Face and code on GitHub, providing a reusable evaluation protocol and scoring via QA accuracy.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce RuleWorld, a large-scale benchmark that reformulates rules as globally reusable abstract units"},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 Findings","evidence":"Accepted by EMNLP 2026 Findings","evidenceUrl":"http://arxiv.org/abs/2608.22753v1","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-25T16:21:05.134052Z"},"venueAttempts":[{"venueName":"EMNLP 2026 Findings","reviewStatus":"accepted","decisionRaw":"Accepted by EMNLP 2026 Findings","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"http://arxiv.org/abs/2608.22753v1","observedAt":"2026-08-25T16:21:05.134052Z","rawValue":"Accepted by EMNLP 2026 Findings","level":"author-claim"}]}],"attentionForecast":{"score":62,"confidence":"Medium","horizon":"7d","reason":"Large-scale dataset and a new training framework with strong reported gains are likely to attract attention in the NLP community."},"evaluationMode":"public_reusable","publishers":[{"name":"RuleWorld contributors","organizationType":"academic-lab","sourceUrl":"https://github.com/SharkSpicy-NLP/Beyond-Factual-Knowledge","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_fff60a0377e326e1","familyId":"catalog_family_fff60a0377e326e1","name":"RuneScape-Bench","oneLine":"An agentic coding benchmark where models use a TypeScript SDK to play a RuneScape-like environment and optimize skill-training performance.","description":"An agentic coding benchmark where models use a TypeScript SDK to play a RuneScape-like environment and optimize skill-training performance.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://maxbittker.github.io/runebench/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fff60a0377e326e1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/runescapebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"runescapeBench","url":"https://benchlm.ai/benchmarks/runescapebench","paperUrl":"https://maxbittker.github.io/runebench/","year":"2026","fullName":"RuneBench / runescape-bench","format":"Average log XP-rate score","tasks":"16 RuneScape skill-training tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_rusfinchain_4bba617f","familyId":"bmf_d05efa853bd6","name":"RusFinChain","oneLine":"RusFinChain is a Russian-language benchmark for verifiable chain-of-thought reasoning in finance, comprising 5,280 parameterized examples generated from executable Python templates across 17 domains and 172 topics. Each example includes a gold-standard reasoning chain with intermediate numeric values for automatic verification, and the dataset, code, and evaluation framework are publicly released.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.01388","pdf":"https://arxiv.org/pdf/2607.01388","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.01388"},"evidence":{"snippet":"We present RusFinChain, the first Russian-language symbolic benchmark for verifiable CoT reasoning in finance.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01388"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"RusFinChain is a Russian-language benchmark for verifiable chain-of-thought reasoning in finance, comprising 5,280 parameterized examples generated from executable Python templates across 17 domains and 172 topics. Each example includes a gold-standard reasoning chain with intermediate numeric values for automatic verification, and the dataset, code, and evaluation framework are publicly released.","whyItMatters":"Most financial reasoning benchmarks lack step-level supervision and are English-only; RusFinChain addresses this gap by providing a contamination-free, verifiable CoT benchmark for Russian, enabling automatic assessment of intermediate reasoning steps. It introduces improved metrics that better correlate with final-answer correctness, supporting more diagnostic evaluation of financial LLMs for the Russian-speaking community.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"409d6d6fd0e6d809175c46b08e44cae85ec1fd5294d22ea4caf8a390b53dfdea"},"motivation":"Multi-step symbolic reasoning is essential for robust financial analysis, yet most benchmarks neglect intermediate reasoning steps.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01388","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_russian-it-community-corpus_d8ce4257","familyId":"bmf_399000b287dd","name":"Russian IT Community Corpus","oneLine":"A corpus of Russian IT community discussions spanning 2017-2026, with splits for SFT dialogues, DPO pairs, and RAG knowledge base chunks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-24","recognitionConfidence":0.4,"links":{"report":"https://github.com/wwewtech/russian-it-community-corpus","pdf":null,"project":"https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-Dataset-FFD21E?style=flat-square","code":"https://github.com/wwewtech/russian-it-community-corpus","data":"https://huggingface.co/datasets/wwewtech/russian-it-community-corpus","hfPaper":null},"evidence":{"snippet":"russian-it-community-corpus Russian IT Community Conversational Corpus (2018-2026) · Zero-PII Curation Platform, Multi-turn SFT, DPO, RAG Knowledge Base, Streamlit Studio & RTX 3060 LoRA benchmark data-engineering dataset dpo llm lora machine-learning rag russian-nlp sft streamlit zero-pii **High-throughput data engineering and Zero-PII curation platform for language models** 2.91M+ discussions · 2017–2026 history · SFT dialogues · DPO pairs · RAG knowledge base · LoRA on RTX 3060 [![Hugging Fac","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":162,"hfDatasetLikes":1},"source":{"type":"github","id":"github:wwewtech/russian-it-community-corpus"},"ranking":{"30d":{"score":23,"rank":105,"coverage":0.7,"confidence":"Medium","datasetDownloadRank":17,"datasetRankPopulation":30},"90d":{"score":20,"rank":387,"coverage":0.85,"confidence":"High","datasetDownloadRank":42,"datasetRankPopulation":66}},"description":"A corpus of Russian IT community discussions spanning 2017-2026, with splits for SFT dialogues, DPO pairs, and RAG knowledge base chunks.","whyItMatters":"The corpus provides a large, de-identified conversational resource for language model training and retrieval tasks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"d552d55c828e8c3be1b64ee401a83aa5d08c9671166d115c1e67ceede2e45eb8"},"motivation":"russian-it-community-corpus Russian IT Community Conversational Corpus (2018-2026) · Zero-PII Curation Platform, Multi-turn SFT, DPO, RAG Knowledge Base, Streamlit Studio & RTX 3060 LoRA benchmark data-engineering dataset dpo llm lora machine-learning rag russian-nlp sft streamlit zero-pii **High-throughput data engineering and Zero-PII curation platform for language models** 2.91M+ discussions · 2017–2026 history ·…","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The repository presents a dataset and data engineering stack, not a benchmark with an explicit scoring contract or submission procedure. No formal benchmark release is identified."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/wwewtech/russian-it-community-corpus","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-24T07:40:52.396154Z"},"attentionForecast":{"score":63,"confidence":"Low","horizon":"7d","reason":"The niche Russian IT domain and lack of a benchmark scoring framework limit broad appeal, though the large dataset and active curation may draw attention."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_rut-bench_450993d2","familyId":"bmf_85a551e57aea","name":"RUT-Bench","oneLine":"RUT-Bench evaluates LLM tool-use in realistic user interactions with 1,638 test samples across 59 executable environments, measuring success rate, informational honesty, and tool discipline.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.03318","pdf":"https://arxiv.org/pdf/2606.03318","project":null,"code":"https://github.com/Miaow-Lab/RUT-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.03318"},"evidence":{"snippet":"To fill this gap, we propose RUT-Bench, a dedicated benchmark designed to assess LLMs under diverse Real-world User Tool calling scenarios.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03318"},"ranking":{"90d":{"score":25,"rank":300,"coverage":0.7,"confidence":"Medium"}},"description":"RUT-Bench evaluates LLM tool-use in realistic user interactions with 1,638 test samples across 59 executable environments, measuring success rate, informational honesty, and tool discipline.","whyItMatters":"Fill the gap of real-world tool-calling evaluation by simulating non-ideal user behaviors, providing a standardized framework for assessing LLM robustness in practical scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"766ccf7f34b8bcb3e7e183803401b6f74fd5906f8bc422861e2e30d391e44643"},"motivation":"Despite great advances in tool-use capabilities of large language models (LLMs), existing evaluation benchmarks struggle to fully align with real-world scenarios.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03318","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Miaow-Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/Miaow-Lab/RUT-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ruverbench_289e3a74","familyId":"bmf_646ce3fcec4f","name":"RuVerBench","oneLine":"RuVerBench evaluates LLM-as-a-judge reliability for rubric verification in agentic scenarios. It includes 2,458 instances across deep research and agentic coding, each with a model-generated output, a rubric, and a human-annotated label indicating rubric satisfaction.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29920","pdf":"https://arxiv.org/pdf/2606.29920","project":null,"code":"https://github.com/THU-KEG/RuVerBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.29920"},"evidence":{"snippet":"We introduce RuVerBench, the first benchmark for assessing LaaJ reliability in rubric verification for agentic scenarios.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":10,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29920"},"ranking":{"90d":{"score":35,"rank":196,"coverage":0.7,"confidence":"Medium"}},"description":"RuVerBench evaluates LLM-as-a-judge reliability for rubric verification in agentic scenarios. It includes 2,458 instances across deep research and agentic coding, each with a model-generated output, a rubric, and a human-annotated label indicating rubric satisfaction.","whyItMatters":"Rubric-based scoring with LLM judges is common but under-validated, especially for agentic outputs. RuVerBench provides a reusable benchmark to compare judge models and strategies, enabling decisions on model selection and scoring protocol.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"20987ca85eee3fb2da929edc4f10fd99d9ae46bfb1b19dfa6a19e2dd7b656988"},"motivation":"Rubric-based scoring has become a widely used paradigm in model evaluation, typically with LLM-as-a-Judge (LaaJ) for rubric scoring.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.29920","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Tsinghua University","organizationType":"academic-lab","sourceUrl":"https://github.com/THU-KEG/RuVerBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_rw-voice-eq-bench_67940c5c","familyId":"bmf_19d351184c25","name":"RW-Voice-EQ Bench","oneLine":"The Real World Voice EQ Bench evaluates voice AI systems across TTS, STS, SU, and ASR, focusing on how well models use acoustic information beyond text. It assesses dimensions like naturalness, expressiveness, identity stability, reliability, vocal affect use, and robustness to real-world conditions such as accent, emotion, noise, and conversation.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SD"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14846","pdf":"https://arxiv.org/pdf/2607.14846","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.14846"},"evidence":{"snippet":"To this end, we introduce the Real World Voice EQ Bench, a multidimensional benchmark for evaluating voice AI across text-to-speech (TTS), speech-to-speech (STS), speech understanding (SU), and automatic speech recognition (ASR).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":11,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14846"},"ranking":{"90d":{"score":49,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"The Real World Voice EQ Bench evaluates voice AI systems across TTS, STS, SU, and ASR, focusing on how well models use acoustic information beyond text. It assesses dimensions like naturalness, expressiveness, identity stability, reliability, vocal affect use, and robustness to real-world conditions such as accent, emotion, noise, and conversation.","whyItMatters":"Current voice AI benchmarks often evaluate isolated capabilities like word error rate or text-based dialogue quality, missing how systems harness acoustic information central to spoken language. This benchmark highlights that performance varies across dimensions, showing that a single aggregate score is insufficient and that real-world conditions expose failures not captured by clean-speech tests. It supports more nuanced evaluation and improvement of voice AI systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ab5af34bf14d8c9c9d952b5298f5eb8f486506392ed9d8dc6ac6f83dbcc56f99"},"motivation":"Current voice AI benchmarks typically evaluate isolated capabilities such as speech intelligibility, word error rate, or text-based dialogue quality, but they rarely test whether systems harness the acoustic information that distinguishes spoken language from its textual representation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14846","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_rwgbench_ed2b1ef6","familyId":"bmf_9188a660e3ea","name":"RWGBench","oneLine":"Evaluates related work generation as citation-centric scholarly positioning. Uses 100 peer-reviewed papers and a 1.09M-document retrieval corpus, with metrics for citation selection, contextual appropriateness, organization, and discourse structure.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.DL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24894","pdf":"https://arxiv.org/pdf/2606.24894","project":null,"code":"https://github.com/BFTree/RWGBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.24894"},"evidence":{"snippet":"However, related work writing is fundamentally a citation-level scholarly positioning task: it requires selecting, organizing, and framing prior work to clarify how a target paper relates to, differs from, and contributes beyond existing research.As a result, models may generate coherent and semantically-relevant text while exhibiting academically critical failures, such as inappropriate citation selection or misplaced references, that conventional metrics do not capture.To this end, we introduc","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24894"},"ranking":{},"description":"Evaluates related work generation as citation-centric scholarly positioning. Uses 100 peer-reviewed papers and a 1.09M-document retrieval corpus, with metrics for citation selection, contextual appropriateness, organization, and discourse structure.","whyItMatters":"Fills the gap in RWG evaluation that relies on surface text similarity, offering a citation-centric testbed that aligns with expert judgment and reveals systematic limitations in current systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a1ecf362651ed4c768e7728026c03a125a995cfb2be63e634041548dbc9f88dc"},"motivation":"Large language models have shown strong fluency in scientific writing, yet the evaluation of related work generation (RWG) remains limited.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24894","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_s2m-bench_0da86db5","familyId":"bmf_9b8a17f82d4d","name":"S2M-Bench","oneLine":"S2M-Bench evaluates the task of reconstructing mind maps from lecture slides, comprising 12,774 slide pages from 24 university courses with expert-annotated mind maps. The evaluation framework integrates ground-truth comparison, structure conformity analysis, and VLM-as-a-Judge scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00610","pdf":"https://arxiv.org/pdf/2608.00610","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00610"},"evidence":{"snippet":"For systematic evaluation, we introduce S2M-Bench, a benchmark comprising 12,774 slide pages with expert-annotated mind maps spanning 24 university courses.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00610"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"S2M-Bench evaluates the task of reconstructing mind maps from lecture slides, comprising 12,774 slide pages from 24 university courses with expert-annotated mind maps. The evaluation framework integrates ground-truth comparison, structure conformity analysis, and VLM-as-a-Judge scoring.","whyItMatters":"Existing benchmarks do not address automatic generation and evaluation of mind maps from educational slides, a task requiring balance of local and global knowledge. S2M-Bench provides a systematic evaluation framework to advance intelligent education tools by enabling comparison of models on this task.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"91cfac840ab0ae0681fc5ea2f9c1950c0568e6755e31d5bf72f999511b32587b"},"motivation":"Generating mind maps from lecture slides can help learners efficiently assimilate fragmented knowledge, promising substantial benefits for intelligent education.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00610","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_s2r-bench_847c5139","familyId":"bmf_912689c7de1c","name":"S2R-Bench","oneLine":"S2R-Bench evaluates video reflection removal, supporting full-reference and human perceptual assessment, and is used to validate the S2R-Removal model.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.11562","pdf":"https://arxiv.org/pdf/2608.11562","project":"https://codingwzp.github.io/VideoDereflection_S2R","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.11562"},"evidence":{"snippet":"We further build S2R-Bench, the first benchmark for video reflection removal, supporting both full-reference evaluation and real-world human perceptual assessment.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":10,"hfDailySubmittedAt":"2026-08-13T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11562"},"ranking":{"30d":{"score":48,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":49,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"S2R-Bench evaluates video reflection removal, supporting full-reference and human perceptual assessment, and is used to validate the S2R-Removal model.","whyItMatters":"Addresses the lack of benchmarks for video reflection removal, providing a dedicated evaluation protocol that includes both synthetic and real-world scenarios, which is crucial for advancing this under-explored task.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e02a15f244a3753241998f5d1a4998b30b8f4debed4f6ae1f24eb39d95245803"},"motivation":"Videos captured through glass often contain reflections that degrade visual quality and interfere with downstream vision tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11562","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_sa-bench_734de9b3","familyId":"bmf_4562cb20b278","name":"SA-Bench","oneLine":"SA-Bench (SemanticAlign-Bench) evaluates semantic alignment in LLM-based paper reproduction across 30 papers from top conferences. It decomposes paper specifications into atomic verifiable claims (SAUs) and evaluates repositories along four diagnostic dimensions (numerical, methodological, protocol, ordering drift). Includes 1,491 SAUs across five ML domains and evaluates 12 generator configurations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.24252","pdf":"https://arxiv.org/pdf/2608.24252","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce SemanticAlign-Bench(SA-Bench), a diagnostic benchmark covering 30 papers from ICLR, ICML and NeurIPS 2025.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24252"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"today":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SA-Bench (SemanticAlign-Bench) evaluates semantic alignment in LLM-based paper reproduction across 30 papers from top conferences. It decomposes paper specifications into atomic verifiable claims (SAUs) and evaluates repositories along four diagnostic dimensions (numerical, methodological, protocol, ordering drift). Includes 1,491 SAUs across five ML domains and evaluates 12 generator configurations.","whyItMatters":"LLM agents generating code for paper reproduction often produce semantically unfaithful implementations. SA-Bench provides a diagnostic framework to measure semantic drift, revealing that current agents struggle with faithful implementation, guiding development of better scaffolding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"1a6f709d680ba9855c1761a223024fd978e786b6cbd1bd3833d2bd36fc399159"},"motivation":"LLM agents can generate paper reproduction code, yet often produce scientifically unfaithful implementations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally named, has a stable scoring contract via SAU evaluation, and states that the benchmark, annotations, and pipeline are publicly available, though no direct artifact link is included.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce SemanticAlign-Bench(SA-Bench), a diagnostic benchmark"},"publication":{"status":"acceptance_claimed","venue":"Findings of EMNLP 2026","evidence":"Accepted to Findings of EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2608.24252","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-26T06:08:22.244942Z"},"venueAttempts":[{"venueName":"Findings of EMNLP 2026","reviewStatus":"accepted","decisionRaw":"Accepted to Findings of EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.24252","observedAt":"2026-08-26T06:08:22.244942Z","rawValue":"Accepted to Findings of EMNLP 2026","level":"author-claim"}]}],"attentionForecast":{"score":60,"confidence":"Medium","horizon":"7d","reason":"Niche but relevant to AI for science and code generation; likely to engage reproducibility-focused researchers."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_saber_ac2e2d88","familyId":"bmf_e4c3c153699b","name":"SABER","oneLine":"SABER evaluates operational safety of LLM coding agents in stateful project workspaces. Agents perform realistic tasks, and safety is scored from the final environment state after action sequences. Violations are categorized by cause, enabling model-specific safety profile analysis.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01317","pdf":"https://arxiv.org/pdf/2606.01317","project":null,"code":"https://github.com/sssr-lab/saber","data":null,"hfPaper":"https://huggingface.co/papers/2606.01317"},"evidence":{"snippet":"We present SABER, a benchmark for environment-aware operational safety that places models in realistic agent-style projects and evaluates safety from the final environment state after a sequence of actions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01317"},"ranking":{},"description":"SABER evaluates operational safety of LLM coding agents in stateful project workspaces. Agents perform realistic tasks, and safety is scored from the final environment state after action sequences. Violations are categorized by cause, enabling model-specific safety profile analysis.","whyItMatters":"Existing safety benchmarks only check prompt refusal, missing the impact of action sequences on workspaces. SABER fills this gap by measuring environment-aware operational safety, offering a practical way to compare models on their ability to avoid harmful state changes in realistic coding tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1e421050f57d308d4abc8168731c709ad0782aec2be5df5be06bd3189bfd7a0e"},"motivation":"Large language models are increasingly deployed as coding agents, shifting safety from individual responses to action sequences.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01317","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SSSR Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/sssr-lab/saber","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_saber-math_7bfa1c5b","familyId":"bmf_7856a310f2fa","name":"SABER-Math","oneLine":"SABER-Math evaluates information retrieval for mathematical queries, with about 283K problems and tasks for reranking based on fine-grained relevance. Scoring uses preference tournament ratings.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.IR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29894","pdf":"https://arxiv.org/pdf/2606.29894","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.29894"},"evidence":{"snippet":"We address this gap by introducing SABER-Math, the first fully automated benchmark for evaluating mathematical IR without expert annotation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29894"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SABER-Math evaluates information retrieval for mathematical queries, with about 283K problems and tasks for reranking based on fine-grained relevance. Scoring uses preference tournament ratings.","whyItMatters":"Existing IR benchmarks fail to capture mathematical relevance, and MTEB doesn't predict math performance. SABER-Math provides a math-specific benchmark to guide retriever selection.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"547f8647ce984c70f7184dd5f84253559a73559fe639920ee0986bfac8515a8f"},"motivation":"As agentic AI systems tackle more complex mathematical tasks, they increasingly rely on information retrieval (IR) to search problem databases, theorem libraries, and educational resources.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"in the 3rd AI for Math Workshop at the 43rd International Conference on Machine Learning (ICML), Seoul, South Korea, 202","evidence":"Accepted in the 3rd AI for Math Workshop at the 43rd International Conference on Machine Learning (ICML), Seoul, South Korea, 2026","evidenceUrl":"https://arxiv.org/abs/2606.29894","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"in the 3rd AI for Math Workshop at the 43rd International Conference on Machine Learning (ICML), Seoul, South Korea, 202","reviewStatus":"accepted","decisionRaw":"Accepted in the 3rd AI for Math Workshop at the 43rd International Conference on Machine Learning (ICML), Seoul, South Korea, 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.29894","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted in the 3rd AI for Math Workshop at the 43rd International Conference on Machine Learning (ICML), Seoul, South Korea, 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval","Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_safebuild-bench_538eb6ea","familyId":"bmf_e22800c63b0a","name":"SafeBuild-Bench","oneLine":"A benchmark for evaluating multimodal large language models on construction safety hazard identification and description, with 3,314 expert-verified task instances from over 3,000 images.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Safety"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00068","pdf":"https://arxiv.org/pdf/2608.00068","project":null,"code":"https://github.com/safebuild/gems","data":null,"hfPaper":"https://huggingface.co/papers/2608.00068"},"evidence":{"snippet":"We introduce SafeBuild-Bench, a metadata-driven benchmark for evaluating multimodal large language models on construction safety under realistic temporal and site variation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00068"},"ranking":{"90d":{"score":25,"rank":297,"coverage":0.7,"confidence":"Medium"}},"description":"A benchmark for evaluating multimodal large language models on construction safety hazard identification and description, with 3,314 expert-verified task instances from over 3,000 images.","whyItMatters":"Construction-safety models must handle realistic temporal and site variation; this benchmark provides a standard evaluation for hazard recognition and description, useful for deployment risk assessment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8131bb22ab159a7d073ef4a03c6739f0a5bc5ebf8348ea3d6b241631f5309bb0"},"motivation":"Construction-safety models must handle concrete deployment risks, such as a worker standing near a scaffold edge without guardrails, rather than only recognize common objects in curated images.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"KDD 2026","evidence":"Accepted by KDD 2026. 12 pages, 6 figures","evidenceUrl":"https://arxiv.org/abs/2608.00068","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"KDD 2026","reviewStatus":"accepted","decisionRaw":"Accepted by KDD 2026. 12 pages, 6 figures","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.00068","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by KDD 2026. 12 pages, 6 figures","level":"author-claim"}]}],"publishers":[{"name":"SafeBuild","organizationType":"academic-lab","sourceUrl":"https://github.com/safebuild/gems","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_safeclawbench_e7a55be8","familyId":"bmf_91d28b9baddb","name":"SafeClawBench","oneLine":"SafeClawBench is a staged benchmark for tool-using LLM agent security with 600 adversarial tasks across six attack families, reporting three endpoints: semantic attack acceptance, audit-visible harm evidence, and sandbox-observed tool/state harm.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18356","pdf":"https://arxiv.org/pdf/2606.18356","project":null,"code":null,"data":"https://huggingface.co/datasets/sairights/safeclawbench","hfPaper":"https://huggingface.co/papers/2606.18356"},"evidence":{"snippet":"We introduce SafeClawBench, a staged benchmark for tool-using agent security with 600 controlled adversarial tasks across six attack families: direct and indirect prompt injection, tool-return injection, memory poisoning, memory extraction, and ambiguity-driven unsafe inference.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":281,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2606.18356"},"ranking":{"90d":{"score":49,"rank":71,"coverage":0.3,"confidence":"Low","datasetDownloadRank":32,"datasetRankPopulation":66}},"description":"SafeClawBench is a staged benchmark for tool-using LLM agent security with 600 adversarial tasks across six attack families, reporting three endpoints: semantic attack acceptance, audit-visible harm evidence, and sandbox-observed tool/state harm.","whyItMatters":"SafeClawBench addresses the evaluation gap where existing benchmarks collapse distinct security failure stages into a single metric, making it hard to distinguish semantic compliance from actual harm. It provides separate, comparable scores across models and prompt policies, aiding in selecting agents and defenses based on the specific type of security risk.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1e8955f32543d95d9f59dfad38699b2ed8ce8a462bccecdf4748be32ffca13c4"},"motivation":"Tool-using language-model agents introduce security failures that go beyond unsafe text: they can disclose protected objects, write persistent memory, send messages, modify databases, or trigger harmful code and tool effects.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18356","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SAIRights","organizationType":"company-research-lab","sourceUrl":"https://huggingface.co/datasets/sairights/safeclawbench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_safegen-bench_15e57f70","familyId":"bmf_dca094d6bfad","name":"SafeGen-Bench","oneLine":"Evaluates safety of conditional text-to-video generation using selected start frames and text prompts across 10 malicious categories, measuring unsafety scores and guardrail effectiveness.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.01481","pdf":"https://arxiv.org/pdf/2606.01481","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01481"},"evidence":{"snippet":"To bridge this gap, we introduce SafeGen-Bench, a benchmark specifically designed to evaluate the safety of conditional T2V models.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01481"},"ranking":{},"description":"Evaluates safety of conditional text-to-video generation using selected start frames and text prompts across 10 malicious categories, measuring unsafety scores and guardrail effectiveness.","whyItMatters":"Addresses the gap of safety evaluation when both text and image inputs are benign but output is harmful. Provides a benchmark to improve model safeguards in dynamic video generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"49841dd9018d5cac29fb4840d4e1bb32d38dfc9217d923b3c86c393b51e76e09"},"motivation":"With the rapid advancements in text-to-image diffusion models, generative video models (T2V models) like Sora can now produce short synthetic videos from a text prompt or an initial image.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01481","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SafeGen-Bench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.01481","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_safegesture_6fcb400d","familyId":"bmf_4d0bd9e289c6","name":"SafeGesture","oneLine":"SafeGesture evaluates vision-language models on scenario-conditioned safety interpretation of hand gestures, pairing 6 gestures with 8 scenarios for 4,800 items.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.16081","pdf":"https://arxiv.org/pdf/2608.16081","project":null,"code":"https://github.com/The-Responsible-AI-Initiative/SafeGesture","data":null,"hfPaper":"https://huggingface.co/papers/2608.16081"},"evidence":{"snippet":"We introduce SafeGesture, a benchmark that evaluates whether a model can infer scenario-appropriate safety actions from hand gestures.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.16081"},"ranking":{"30d":{"score":23,"rank":149,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":353,"coverage":0.55,"confidence":"Low"}},"description":"SafeGesture evaluates vision-language models on scenario-conditioned safety interpretation of hand gestures, pairing 6 gestures with 8 scenarios for 4,800 items.","whyItMatters":"It exposes a perception-reasoning gap in safety-critical gesture interpretation, providing a reusable test for scenario-dependent decision-making.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"59114b1de9b4fa9805dd83614bcf8dc7c5b9ab52ec9083f2b1aaf1219b3ef0fc"},"motivation":"Open-weight and frontier vision-language models (VLMs) perform well on general image understanding, but their ability to interpret fine-grained hand gestures in safety-critical operational contexts remains largely unexamined.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.16081","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"The Responsible AI Initiative","organizationType":"benchmark-organization","sourceUrl":"https://github.com/The-Responsible-AI-Initiative/SafeGesture","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_safepyramid_fe239f10","familyId":"bmf_efe4e93686e2","name":"SafePyramid","oneLine":"SafePyramid evaluates in-context policy guardrailing across 1,000 multi-turn conversations and 3,000 application-specific policies containing 61,699 natural-language rules, organized into three hierarchical capability levels (L0, L1, L2). Scoring is based on violated-rule set prediction with metrics RMR and RDR, and the evaluation harness supports any API or local model.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29887","pdf":"https://arxiv.org/pdf/2606.29887","project":null,"code":"https://github.com/bytedance/safepyramid","data":null,"hfPaper":"https://huggingface.co/papers/2606.29887"},"evidence":{"snippet":"To systematically evaluate this capability, we introduce SafePyramid, a safety benchmark comprising 1,000 multi-turn conversations across 10 domains and 3,000 corresponding application-specific policies, which together contain 61,699 distinct natural-language rules.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":6,"hfDailySubmittedAt":"2026-06-30T00:00:00.000Z","githubStars":8,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29887"},"ranking":{"90d":{"score":37,"rank":181,"coverage":0.7,"confidence":"Medium"}},"description":"SafePyramid evaluates in-context policy guardrailing across 1,000 multi-turn conversations and 3,000 application-specific policies containing 61,699 natural-language rules, organized into three hierarchical capability levels (L0, L1, L2). Scoring is based on violated-rule set prediction with metrics RMR and RDR, and the evaluation harness supports any API or local model.","whyItMatters":"SafePyramid addresses the gap in evaluating guardrails under application-specific policies rather than fixed taxonomies, providing a structured test for rule understanding, dependency resolution, and adaptation to novel frameworks. It enables direct comparison of frontier LLMs and configurable guardrails on a realistic safety task.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"29270bd355d906f7bc22eec279f2e1c76f1c24b56f454878457e62a4f07f1159"},"motivation":"In real-world applications, guardrails are often expected to identify unsafe user-model interactions according to application-specific safety policies, rather than relying on predefined risk taxonomies.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.29887","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ByteDance","organizationType":"company-research-lab","sourceUrl":"https://github.com/bytedance/safepyramid","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_saferelbench_483807a4","familyId":"bmf_e7af30443a9d","name":"SafeRelBench","oneLine":"SafeRelBench evaluates VLM-driven embodied agents on household tasks with process-level safety constraints, focusing on spatial relations such as support, containment, and proximity. It includes 507 executable samples (248 spatial-relation, 259 control) and measures whether agents satisfy safety conditions before risk-prone actions, alongside task success.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14543","pdf":"https://arxiv.org/pdf/2607.14543","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.14543"},"evidence":{"snippet":"To address this gap, we introduce SAFERELBENCH, a spatial-relation-aware safety benchmark with 507 executable evaluation samples, including 248 spatial-relation samples and 259 non-spatial control samples.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14543"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SafeRelBench evaluates VLM-driven embodied agents on household tasks with process-level safety constraints, focusing on spatial relations such as support, containment, and proximity. It includes 507 executable samples (248 spatial-relation, 259 control) and measures whether agents satisfy safety conditions before risk-prone actions, alongside task success.","whyItMatters":"Safety in embodied agents depends on spatial awareness during action sequences, not just final outcomes. SafeRelBench fills a gap by quantifying process-level safety compliance, enabling comparison across agents and highlighting the need for improved spatial reasoning in safe planning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"90b9a8458215916985e3beec8e3cda0da201989b6161f7a78a6812b7884e99e6"},"motivation":"Vision-language models (VLMs) are increasingly used as the reasoning backbone of embodied agents, enabling robots to interpret visual scenes, follow language instructions, and plan multi-step actions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14543","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Authors of SafeRelBench","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.14543","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_safescenereason_c358bce3","familyId":"bmf_7f963029c8c3","name":"SafeSceneReason","oneLine":"SafeSceneReason evaluates multimodal industrial-safety reasoning with 123,695 question-answer pairs covering compliance, hazard interaction, accident mechanisms, and prevention recommendations across scene-centric and report-centric pipelines.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Factuality"],"topics":["Multimodal","Safety","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09230","pdf":"https://arxiv.org/pdf/2608.09230","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09230"},"evidence":{"snippet":"We introduce SafeSceneReason, a multimodal industrial-safety reasoning benchmark and companion training corpus that connects workplace scenes with knowledge from occupational accident investigations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09230"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SafeSceneReason evaluates multimodal industrial-safety reasoning with 123,695 question-answer pairs covering compliance, hazard interaction, accident mechanisms, and prevention recommendations across scene-centric and report-centric pipelines.","whyItMatters":"Existing safety datasets test perception or isolated violations, leaving a gap in evidence-grounded reasoning. This benchmark differentiates model competence in comparative, technical, and multi-evidence reasoning, offering decision value for industrial safety applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2581022515208fffc9aeb20f9f67a4dd65299dad78eb9320defe140fad1729e2"},"motivation":"Industrial-safety understanding requires more than detecting workers, equipment, and personal protective equipment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09230","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_sagaqa_9735b00e","familyId":"bmf_b53a4f200da8","name":"SagaQA","oneLine":"SagaQA is a long-form video benchmark for multi-hop reasoning over full-length TV series, requiring reasoning across episodes and multimodal narrative understanding.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03301","pdf":"https://arxiv.org/pdf/2606.03301","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03301"},"evidence":{"snippet":"We introduce SagaQA, a long-form video benchmark for multi-hop reasoning over full-length TV series.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03301"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SagaQA is a long-form video benchmark for multi-hop reasoning over full-length TV series, requiring reasoning across episodes and multimodal narrative understanding.","whyItMatters":"No public artifacts or reuse path are provided, and the benchmark lacks a clear scoring contract for independent runs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"88c1e60cba1a408c6971f2b08c0b72159ad91455bd2ba861d227db2d78a97d2c"},"motivation":"We introduce SagaQA, a long-form video benchmark for multi-hop reasoning over full-length TV series.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03301","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_7e5d4325a44714fc","familyId":"catalog_family_7e5d4325a44714fc","name":"SAGE","oneLine":"Student Assessment with Generative Evaluation","description":"Student Assessment with Generative Evaluation","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/sage","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7e5d4325a44714fc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/sage"}],"catalogSources":[{"catalog":"benchlm","sourceId":"sage","url":"https://benchlm.ai/benchmarks/sage","paperUrl":"https://www.vals.ai/benchmarks/sage","year":"2026","fullName":"Vals SAGE","format":"Accuracy score","tasks":"Student assessment with generative evaluation","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_salart-vqa_f96d6eb3","familyId":"bmf_22505c6b720b","name":"SalArt-VQA","oneLine":"SalArt-VQA is a closed-set visual question answering benchmark for fine-grained salient artifact understanding in AI-generated images. It includes 950 images and 3,681 human-authored multiple-choice questions covering artifact images, matched real references, and paired generated references. Four aligned question types evaluate presence detection, semantic localization, spatial grounding, and evidence-grounded defect identification, with reference splits for calibration and abstention testing.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.12671","pdf":"https://arxiv.org/pdf/2606.12671","project":null,"code":null,"data":"https://huggingface.co/datasets/salartvqa/SalArt-VQA","hfPaper":"https://huggingface.co/papers/2606.12671"},"evidence":{"snippet":"To evaluate these behaviors directly, we introduce SalArt-VQA, a diagnostic benchmark for fine-grained SALient ARTifact understanding in AI-generated images.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":131,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2606.12671"},"ranking":{"90d":{"score":41,"rank":138,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":44,"datasetRankPopulation":66}},"description":"SalArt-VQA is a closed-set visual question answering benchmark for fine-grained salient artifact understanding in AI-generated images. It includes 950 images and 3,681 human-authored multiple-choice questions covering artifact images, matched real references, and paired generated references. Four aligned question types evaluate presence detection, semantic localization, spatial grounding, and evidence-grounded defect identification, with reference splits for calibration and abstention testing.","whyItMatters":"Image-level artifact detection accuracy can conceal failures in grounding and evidence use. This benchmark provides a fine-grained evaluation protocol that isolates specific failure modes, enabling comparison of VLMs on their ability to support artifact claims with local visual evidence. It offers practical value for developers selecting models for trustworthy artifact analysis in generated image workflows.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"86c84fb08aa08659e330506ba3665d09266cff50f02d60d2a9f0bf8f8c6e6408"},"motivation":"Vision-language models (VLMs) are increasingly used to detect whether AI-generated images contain visible artifacts, yet their ability to analyze such artifacts remains poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.12671","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SalArt-VQA Benchmark Team","organizationType":"community","sourceUrl":"https://huggingface.co/datasets/salartvqa/SalArt-VQA","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_05a3b0ef55120ee4","familyId":"catalog_family_05a3b0ef55120ee4","name":"SAT Math","oneLine":"SAT Math benchmark from AGIEval containing standardized mathematics questions from the College Board SAT examination, designed to evaluate mathematical reasoning capabilities of foundation models using human-centric assessment methods.","description":"SAT Math benchmark from AGIEval containing standardized mathematics questions from the College Board SAT examination, designed to evaluate mathematical reasoning capabilities of foundation models using human-centric assessment methods.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/sat-math","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_05a3b0ef55120ee4"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/sat-math"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"sat-math","url":"https://llm-stats.com/benchmarks/sat-math","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_sbmllm-bench_68cf829e","familyId":"bmf_33e4d26b1444","name":"SBMLLM-Bench","oneLine":"Evaluates LLM reconstruction of executable systems-biology models from scientific papers using metrics like simulation ratio, species/reaction recovery, stoichiometric error, and AAFE reproducibility.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/cosbi-research/SBMLLM-Bench","pdf":null,"project":"https://www.cosbi.eu/contact","code":"https://github.com/cosbi-research/SBMLLM-Bench","data":null,"hfPaper":null},"evidence":{"snippet":"SBMLLM-Bench Benchmark LLMs on their ability to reconstruct executable systems-biology models from scientific papers.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:cosbi-research/sbmllm-bench"},"ranking":{"30d":{"score":23,"rank":141,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":345,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates LLM reconstruction of executable systems-biology models from scientific papers using metrics like simulation ratio, species/reaction recovery, stoichiometric error, and AAFE reproducibility.","whyItMatters":"Addresses the gap in assessing whether LLMs can accurately convert published biological evidence into executable models, providing a reusable dataset and metrics for model quality.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"28bdb946d1cd7754d77b67f626a8be8becc8cc457e5076ee0ebe10d149da9b55"},"motivation":"SBMLLM-Bench Benchmark LLMs on their ability to reconstruct executable systems-biology models from scientific papers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/cosbi-research/SBMLLM-Bench","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"attentionForecast":{"score":15,"confidence":"Low","horizon":"7d","reason":"Niche systems biology domain with limited audience, but the benchmark has clear artifacts and potential interest from specialized researchers."},"evaluationMode":"public_reusable","publishers":[{"name":"COSBI","organizationType":"academic-lab","sourceUrl":"https://github.com/cosbi-research/SBMLLM-Bench","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_sc3_888d1aed","familyId":"bmf_0092e4d9b090","name":"SC3","oneLine":"SC3 is a multi-solvent solubility benchmark built on BigSolDB v2.1 with 101,535 measurements over 1,327 solutes and 206 solvents. It features a reproducible curation pipeline, nested Gold/Silver/Bronze consensus tiers, leakage-checked splits, and a multi-solvent metric suite (PS-RMSE, Z-RMSE).","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals"],"capabilities":[],"topics":["physics.chem-ph"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07656","pdf":"https://arxiv.org/pdf/2606.07656","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07656"},"evidence":{"snippet":"We introduce SC3, a multi-solvent solubility benchmark built on BigSolDB v2.1 with three contributions: (i) a reproducible curation pipeline yielding 101,535 measurements over 1,327 solutes and 206 solvents, with a recalibrated aleatoric floor of 0.106 log S-roughly 6 times tighter than the conventional figure; (ii) nested Gold/Silver/Bronze consensus tiers with per-point standard deviation, three leakage-checked splits, and a multi-solvent metric suite (PS-RMSE, Z-RMSE); and (iii) a 31-model be","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07656"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SC3 is a multi-solvent solubility benchmark built on BigSolDB v2.1 with 101,535 measurements over 1,327 solutes and 206 solvents. It features a reproducible curation pipeline, nested Gold/Silver/Bronze consensus tiers, leakage-checked splits, and a multi-solvent metric suite (PS-RMSE, Z-RMSE).","whyItMatters":"Existing solubility benchmarks vary in curation and evaluation, hiding model failures. SC3 provides a standardized, reproducible benchmark with calibrated aleatoric limits to better measure model performance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"699d4e94c5fb06c7303497a102ea178d22c935b839af3bdbe7d253c8b04b5180"},"motivation":"Solubility prediction is a standard benchmark in computational chemistry, yet multi-solvent models which reportedly approach the experimental-noise ceiling (i.e.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07656","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_reconstructing-the-right-episode-evaluatin_98f96c99","familyId":"bmf_c199dc017445","name":"SCALE-QA","oneLine":"SCALE-QA evaluates interleaved conversational memory using 3,000 audited multiple-choice questions across 10 domains, where correct answers depend on causally related evidence from earlier turns in flat unsegmented threads.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-26","firstSeenAt":"2026-08-27","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.25655","pdf":"https://arxiv.org/pdf/2608.25655","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce SCALE-QA, a constraint-grounded task QA benchmark for flat unsegmented threads targeting episode integrity failure.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25655"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SCALE-QA evaluates interleaved conversational memory using 3,000 audited multiple-choice questions across 10 domains, where correct answers depend on causally related evidence from earlier turns in flat unsegmented threads.","whyItMatters":"It tests episode integrity failure in long, mixed-topic conversations, a harder memory regime than benchmarks with explicit topic boundaries, and provides deterministic grading for comparing QA performance.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"1805af50c05d20ee7e6fe8b74902055190078f6a89a885240da7ce01a950d7fd"},"motivation":"Conversations with chat assistants increasingly span many topics in a single long-running thread, challenging memory systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:12:36.511123Z","model":"deepseek-v4-pro","decisionReason":"The benchmark introduces a named dataset with a clear evaluation object, deterministic four-way multiple-choice grading, and an explicit release of questions and runtime builder, enabling reuse by other teams.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce SCALE-QA, a constraint-grounded task QA benchmark for flat unsegmented threads targeting episode integrity failure."},"publication":{"status":"acceptance_claimed","venue":"Main Conference of EMNLP 2026","evidence":"19 pages, 6 figures, 30 tables. Accepted to the Main Conference of EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2608.25655","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-27T04:12:10.575570Z"},"venueAttempts":[{"venueName":"Main Conference of EMNLP 2026","reviewStatus":"accepted","decisionRaw":"19 pages, 6 figures, 30 tables. Accepted to the Main Conference of EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.25655","observedAt":"2026-08-27T04:12:10.575570Z","rawValue":"19 pages, 6 figures, 30 tables. Accepted to the Main Conference of EMNLP 2026","level":"author-claim"}]}],"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"The arXiv paper describes a concrete benchmark with 3,000 questions, deterministic scoring, and strong baseline comparisons across multiple LLMs, which is likely to attract moderate attention among NLP researchers working on long-context memory."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning","Long Context & Memory"],"domainScope":"general"},{"id":"bm_scbench-long_49d2490a","familyId":"bmf_8fb55cb2c14b","name":"scBench-Long","oneLine":"scBench-Long evaluates long-horizon single-cell biology reasoning. Agents must recover scientific conclusions from raw or near-raw data without prescribed methods. It contains 21 evaluations spanning diverse biological contexts, with deterministic grading and trajectory rubrics.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["q-bio.GN"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26563","pdf":"https://arxiv.org/pdf/2606.26563","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.26563"},"evidence":{"snippet":"We introduce scBench-Long, a benchmark for long-horizon single-cell biology in which agents must recover scientific conclusions from raw or near-raw data without prescribed methods.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26563"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"scBench-Long evaluates long-horizon single-cell biology reasoning. Agents must recover scientific conclusions from raw or near-raw data without prescribed methods. It contains 21 evaluations spanning diverse biological contexts, with deterministic grading and trajectory rubrics.","whyItMatters":"Existing AI-biology benchmarks measure broad knowledge or local steps; scBench-Long assesses end-to-end scientific claim production. It provides a reusable evaluation for long-horizon reasoning in single-cell data analysis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0ce8128d983bac74942c4aee784ff35a4f35064bc94ddcb88f4b278bcc660cb9"},"motivation":"Single-cell studies require analysts to convert raw measurements into specific biological claims through multi-step workflows and integration of metadata, assay context, and auxiliary evidence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.26563","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_sceneactbench_da42bbe6","familyId":"bmf_9aedebf94723","name":"SceneActBench","oneLine":"A benchmark for visually conditioned action on complete multi-object 3D scenes across five tasks under a unified agent-environment loop, scored against hidden geometric ground truth.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.22393","pdf":"https://arxiv.org/pdf/2607.22393","project":null,"code":"https://github.com/Feinaldo2/SceneActBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.22393"},"evidence":{"snippet":"We present SceneActBench, a benchmark for visually conditioned action across five 3D tasks under a unified agent-environment loop.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":7,"hfDailySubmittedAt":null,"githubStars":11,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.22393"},"ranking":{"90d":{"score":39,"rank":159,"coverage":0.7,"confidence":"Medium"}},"description":"A benchmark for visually conditioned action on complete multi-object 3D scenes across five tasks under a unified agent-environment loop, scored against hidden geometric ground truth.","whyItMatters":"Addresses the under-evaluation of agent action on complete 3D scenes, providing a comparative basis for VLM agents acting on 3D environments.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"71c1287911aba64f84386cf017f29eb6f0743eb3d03f06eeaed0bc8953728f45"},"motivation":"Vision-language model (VLM) agents increasingly use tools to act on 3D scenes rather than only describe them.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.22393","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_schedbench_f5e0e993","familyId":"bmf_e3d434ea145f","name":"SCHEDBench","oneLine":"SCHEDBench evaluates LLM constraint faithfulness in natural-language combinatorial scheduling. It includes 1,132 instances across JSP, RCPSP, nurse rostering, and timetabling, with templated NL variations and solver-verified references.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00991","pdf":"https://arxiv.org/pdf/2608.00991","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00991"},"evidence":{"snippet":"This paper introduces SCHEDBench, a natural-language benchmark for evaluating combinatorial scheduling constraint faithfulness under surface-form variation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00991"},"ranking":{"30d":{"score":41,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":45,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SCHEDBench evaluates LLM constraint faithfulness in natural-language combinatorial scheduling. It includes 1,132 instances across JSP, RCPSP, nurse rostering, and timetabling, with templated NL variations and solver-verified references.","whyItMatters":"LLMs must generate constraint-feasible schedules under surface-form variations. SCHEDBench provides a benchmark to test invariance to paraphrasing, revealing faithfulness issues and informing robust model development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9910dd16df5bc57ab08142d605f873c335b7e6267cc79a7d344e10615d4063f4"},"motivation":"This paper introduces SCHEDBench, a natural-language benchmark for evaluating combinatorial scheduling constraint faithfulness under surface-form variation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00991","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_schemagui_ec6e8d66","familyId":"bmf_3e223257e5f4","name":"SchemaGUI","oneLine":"Evaluates controllable GUI generation via deterministic function-call references synthesized from parameterized schemas across six bilingual scenarios.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-23","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.22390v1","pdf":"https://arxiv.org/pdf/2608.22390v1","project":null,"code":"https://github.com/xdong2002/SchemaGUI","data":null,"hfPaper":null},"evidence":{"snippet":"To address this, we propose SchemaGUI, a template-based benchmark for controllable GUI generation evaluation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22390"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates controllable GUI generation via deterministic function-call references synthesized from parameterized schemas across six bilingual scenarios.","whyItMatters":"Offers scalable, annotation-free evaluation for GUI generation and reveals bottlenecks in geometric spatial control among LLMs.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"c720dc0cefaf6df6751b6309bb52801b5ff5e4ec0bdbb731ec1b68950918dcc9"},"motivation":"Large language models (LLMs) have demonstrated strong potential in graphical user interface (GUI) generation, but reliable evaluation remains challenging due to uncontrolled data distributions, noisy annotations, and limited layout scenario coverage.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with a template-based protocol and code repository, though the code status is unverified.","canonicalNameSource":"abstract","canonicalNameEvidence":"we propose SchemaGUI, a template-based benchmark for controllable GUI generation evaluation"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22390v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":48,"confidence":"Low","horizon":"7d","reason":"Practical GUI generation evaluation with bilingual coverage may draw moderate interest, but unverified code link limits confidence."},"evaluationMode":"public_reusable","publishers":[{"name":"SchemaGUI contributors","organizationType":"academic-lab","sourceUrl":"https://github.com/xdong2002/SchemaGUI","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_sci-vbench_8bc62acf","familyId":"bmf_063001d7adb8","name":"Sci-VBench","oneLine":"Sci-VBench evaluates text-to-video generation across 1,253 expert-annotated examples in 60 scientific subjects. It requires temporally rich videos demonstrating scientific reasoning and knowledge-grounded synthesis. Scoring covers four dimensions: prompt grounding, scientific correctness, spatiotemporal consistency, and low-level perceptual fidelity, using a rubric-based protocol with MLLM judges.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Factuality"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09873","pdf":"https://arxiv.org/pdf/2608.09873","project":null,"code":"https://github.com/sci-vbench/sci-vbench","data":null,"hfPaper":"https://huggingface.co/papers/2608.09873"},"evidence":{"snippet":"We introduce Sci-VBench, a comprehensive benchmark for evaluating knowledge- and reasoning-intensive video generation across scientific domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":29,"hfDailySubmittedAt":"2026-08-11T00:00:00.000Z","githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09873"},"ranking":{"30d":{"score":39,"rank":51,"coverage":0.85,"confidence":"High"},"90d":{"score":36,"rank":183,"coverage":0.7,"confidence":"Medium"}},"description":"Sci-VBench evaluates text-to-video generation across 1,253 expert-annotated examples in 60 scientific subjects. It requires temporally rich videos demonstrating scientific reasoning and knowledge-grounded synthesis. Scoring covers four dimensions: prompt grounding, scientific correctness, spatiotemporal consistency, and low-level perceptual fidelity, using a rubric-based protocol with MLLM judges.","whyItMatters":"Existing video generation benchmarks focus on surface realism, leaving scientific and causal correctness unmeasured. Sci-VBench provides a public protocol and open dataset to compare models on knowledge-intensive generation, revealing gaps between visual quality and reliable scientific dynamics.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0bea81b4ff8609c221e7040b2848b4fe21dbaa5d03563ab582c5f638e731ee58"},"motivation":"We introduce Sci-VBench, a comprehensive benchmark for evaluating knowledge- and reasoning-intensive video generation across scientific domains.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09873","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_benchmarking-ai-agents-for-addressing-scie_d68cb23d","familyId":"bmf_19391d431090","name":"SciAgentArena","oneLine":"Interactive, agent-agnostic environment with ~200 scientific research tasks evaluated via stepwise verification.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-10","firstSeenAt":"2026-08-19","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2606.12736","pdf":"https://arxiv.org/pdf/2606.12736","project":"https://sciagentarena.github.io/","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Here, we introduce SciAgentArena, a systematic benchmark for evaluating AI agents in real-world scientific research scenarios drawn from emerging needs across multiple domains.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":"2026-06-15T00:00:00.000Z","githubStars":9,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.12736"},"ranking":{"90d":{"score":37,"rank":176,"coverage":0.7,"confidence":"Medium"}},"description":"Interactive, agent-agnostic environment with ~200 scientific research tasks evaluated via stepwise verification.","whyItMatters":"Addresses the lack of interactive evaluation for AI agents in complex, heterogeneous scientific workflows, enabling progress measurement in open-ended research tasks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:38:15.981230Z","inputHash":"2d6a64c61a148aba2c7aee1fbfd07c7c8a30a06171d4d0e97a881e23fddcd809"},"motivation":"AI agents are increasingly being developed to accelerate scientific discovery, yet their practical capabilities in real research settings remain poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T17:38:15.981230Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with stepwise verification, an interactive environment, and a public project link for tasks, datasets, and codes.","canonicalNameSource":"abstract","canonicalNameEvidence":"Here, we introduce SciAgentArena, a systematic benchmark for evaluating AI agents in real-world scientific research scenarios"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.12736","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":55,"confidence":"Low","horizon":"7d","reason":"Topical focus on AI for science and a public leaderboard may attract moderate initial interest, but the scope is experimental and no adoption evidence is supplied."},"evaluationMode":"score_submission","publishers":[{"name":"SciAgentArena Team","organizationType":"academic-lab","sourceUrl":"https://sciagentarena.github.io/","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_45a49a95518a947e","familyId":"catalog_family_45a49a95518a947e","name":"SciCode","oneLine":"SciCode is a research coding benchmark curated by scientists that challenges language models to code solutions for scientific problems. It contains 338 subproblems decomposed from 80 challenging main problems across 16 natural science sub-fields including mathematics, physics, chemistry, biology, and materials science. Problems require knowledge recall, reasoning, and code synthesis skills.","description":"SciCode is a research coding benchmark curated by scientists that challenges language models to code solutions for scientific problems. It contains 338 subproblems decomposed from 80 challenging main problems across 16 natural science sub-fields including mathematics, physics, chemistry, biology, and materials science. Problems require knowledge recall, reasoning, and code synthesis skills.","area":"Code & Software","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Coding","Math","Physics","Reasoning","Biology","Chemistry","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://benchlm.ai/benchmarks/scicode","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_45a49a95518a947e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/scicode"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/scicode"}],"catalogSources":[{"catalog":"benchlm","sourceId":"sciCode","url":"https://benchlm.ai/benchmarks/scicode","paperUrl":null,"year":2024,"fullName":"Scientific Code Benchmark","format":null,"tasks":80,"successorKey":null},{"catalog":"llm-stats","sourceId":"scicode","url":"https://llm-stats.com/benchmarks/scicode","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","math","physics","reasoning","biology","chemistry","code"],"catalogModelCount":21,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"specific"},{"id":"bm_sciconbench_a454d95f","familyId":"bmf_2d99508d3db9","name":"SciConBench","oneLine":"SciConBench evaluates AI agents on open-domain scientific conclusion synthesis using 9,110 questions and expert-written conclusions from systematic reviews. The evaluation decomposes conclusions into atomic facts and measures correctness and comprehensiveness via factual precision and recall.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.11337","pdf":"https://arxiv.org/pdf/2606.11337","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.11337"},"evidence":{"snippet":"We introduce SciConBench, a large-scale live benchmark of 9.11K questions and expert-written conclusions from systematic reviews to evaluate open-domain scientific conclusion synthesis.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.11337"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SciConBench evaluates AI agents on open-domain scientific conclusion synthesis using 9,110 questions and expert-written conclusions from systematic reviews. The evaluation decomposes conclusions into atomic facts and measures correctness and comprehensiveness via factual precision and recall.","whyItMatters":"The benchmark addresses the lack of reliable evaluation for AI agents summarizing scientific evidence in health and other critical fields, providing a way to measure factual accuracy and completeness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"368e8ff0280e8b5296a4fb92a5d1ad53d87c589dc570ca33575e3dca4ccc9010"},"motivation":"Scientific AI agents increasingly retrieve evidence, reason across sources, and synthesize conclusions used in consequential decisions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11337","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_scidraw-bench_755dfcbf","familyId":"bmf_6c84b0b5c404","name":"SciDraw-Bench","oneLine":"SciDraw-Bench is a benchmark for scientific figure generation, with 32 tasks across eight figure types and ten disciplines. Each task pairs a natural-language prompt with a machine-checkable specification, and evaluation uses four dimensions: Text Fidelity, Semantic Correctness, Structural Quality, and Convention Adherence.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.28406","pdf":"https://arxiv.org/pdf/2606.28406","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.28406"},"evidence":{"snippet":"We introduce SciDraw-Bench, a benchmark of 32 structured scientific-figure generation tasks spanning eight figure types and ten disciplines, where each task pairs a natural-language prompt with a machine-checkable specification of required labels, relations, components, conventions, and negative constraints.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.28406"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SciDraw-Bench is a benchmark for scientific figure generation, with 32 tasks across eight figure types and ten disciplines. Each task pairs a natural-language prompt with a machine-checkable specification, and evaluation uses four dimensions: Text Fidelity, Semantic Correctness, Structural Quality, and Convention Adherence.","whyItMatters":"Fills the gap in evaluating scientific figure generation, which requires correct labels and diagrammatic structure beyond natural-image composition. It provides a protocol for measuring usability of generated figures, aiding development of domain-specific generative systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dd1e6f68293f117d665d0b51d23ab571b6b6975f15148dc77a58cb2fe14f239a"},"motivation":"Text-to-image and multimodal generative models are increasingly used to produce scientific figures such as mechanism diagrams, experimental-design schematics, conceptual frameworks, and graphical abstracts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.28406","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_ffc23c64f796b685","familyId":"catalog_family_ffc23c64f796b685","name":"ScienceQA","oneLine":"ScienceQA is the first large-scale multimodal science question answering benchmark with 21,208 multiple-choice questions covering 3 subjects (natural science, language science, social science), 26 topics, 127 categories, and 379 skills. The benchmark includes both text and image modalities, featuring detailed explanations and Chain-of-Thought reasoning to diagnose multi-hop reasoning ability.","description":"ScienceQA is the first large-scale multimodal science question answering benchmark with 21,208 multiple-choice questions covering 3 subjects (natural science, language science, social science), 26 topics, 127 categories, and 379 skills. The benchmark includes both text and image modalities, featuring detailed explanations and Chain-of-Thought reasoning to diagnose multi-hop reasoning ability.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/scienceqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ffc23c64f796b685"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/scienceqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"scienceqa","url":"https://llm-stats.com/benchmarks/scienceqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","multimodal","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_3bf2b0b37bff3180","familyId":"catalog_family_3bf2b0b37bff3180","name":"ScienceQA Visual","oneLine":"ScienceQA Visual is a multimodal science question answering benchmark consisting of 21,208 multiple-choice questions from elementary and high school science curricula. The dataset covers 3 subjects (natural science, language science, social science), 26 topics, 127 categories, and 379 skills. 48.7% of questions include image context requiring multimodal reasoning. Questions are annotated with lectures (83.9%) and explanations (90.5%) to support chain-of-thought reasoning for science question answering.","description":"ScienceQA Visual is a multimodal science question answering benchmark consisting of 21,208 multiple-choice questions from elementary and high school science curricula. The dataset covers 3 subjects (natural science, language science, social science), 26 topics, 127 categories, and 379 skills. 48.7% of questions include image context requiring multimodal reasoning. Questions are annotated with lectures (83.9%) and explanations (90.5%) to support chain-of-thought reasoning for science question answering.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/scienceqa-visual","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3bf2b0b37bff3180"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/scienceqa-visual"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"scienceqa-visual","url":"https://llm-stats.com/benchmarks/scienceqa-visual","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_sciexplore_df25f831","familyId":"bmf_0a205bd9cb5c","name":"SciExplore","oneLine":"SciExplore evaluates scientific information-seeking and reasoning in LLMs and agents with four task types: scientific database navigation, ambiguous literature retrieval, missing reference completion, and cross-source structured knowledge synthesis, spanning 103 expert-curated tasks across more than ten scientific disciplines.","area":"Robotics & Embodied AI","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20926","pdf":"https://arxiv.org/pdf/2607.20926","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20926"},"evidence":{"snippet":"We introduce SciExplore, a benchmark designed to evaluate scientific information-seeking and reasoning capabilities of LLMs and agents.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20926"},"ranking":{"90d":{"score":45,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SciExplore evaluates scientific information-seeking and reasoning in LLMs and agents with four task types: scientific database navigation, ambiguous literature retrieval, missing reference completion, and cross-source structured knowledge synthesis, spanning 103 expert-curated tasks across more than ten scientific disciplines.","whyItMatters":"Existing benchmarks emphasize general-domain retrieval or static QA, leaving a gap in assessing realistic scientific workflows. SciExplore's progressive task complexity provides a way to compare model capabilities in entity-level reasoning, document identification, evidence grounding, and domain-level synthesis, informing deployment in scientific research contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"57d2d7f78bd5ff16fcdce1215b7d6fa62bd3fddeb61418c5fc8340d0e45d5b2b"},"motivation":"Scientific research involves complex information-seeking and reasoning workflows across heterogeneous sources.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20926","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SciExplore Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.20926","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"general"},{"id":"bm_scifigbench_9f0ee596","familyId":"bmf_aefd11b3865b","name":"SciFigBench","oneLine":"SciFigBench is a diagnostic VLM benchmark for scientific figure understanding, covering perception, reasoning, and behavioral reliability under uncertainty.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.13267","pdf":"https://arxiv.org/pdf/2608.13267","project":"https://scifigbench.nlp4sci.com","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.13267"},"evidence":{"snippet":"We introduce SciFigBench, a diagnostic VLM benchmark for scientific figure understanding that jointly evaluates perception, reasoning, and behavioral reliability under uncertainty.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13267"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SciFigBench is a diagnostic VLM benchmark for scientific figure understanding, covering perception, reasoning, and behavioral reliability under uncertainty.","whyItMatters":"It addresses the gap in evaluating VLMs' behavior when visual evidence is missing or misleading, critical for scientific workflows.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ac9e5125c28750f209758db7177029cd0e9805d066722bcf9f5cec57385dc57a"},"motivation":"Existing vision-language model (VLM) benchmarks emphasize perception and reasoning accuracy (how well VLMs describe and reason about what they see in an image), with limited attention to behavioral reliability under uncertainty (how they behave when visual evidence is missing or misleading).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13267","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_scifigplag-bench_b0402713","familyId":"bmf_98ef94f5375b","name":"SciFigPlag-Bench","oneLine":"SciFigPlag-Bench evaluates provenance-aware reasoning for scientific figure plagiarism detection. It includes 2,582 positive and 2,541 negative pairs, with tasks for pairwise detection, source attribution, reuse-type classification, and reuse correspondence localization.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.29124","pdf":"https://arxiv.org/pdf/2607.29124","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.29124"},"evidence":{"snippet":"We present SciFigPlag-Bench, a benchmark for provenance-aware reasoning over scientific figures in scholarly documents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.29124"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SciFigPlag-Bench evaluates provenance-aware reasoning for scientific figure plagiarism detection. It includes 2,582 positive and 2,541 negative pairs, with tasks for pairwise detection, source attribution, reuse-type classification, and reuse correspondence localization.","whyItMatters":"Figure plagiarism is underexplored and general similarity benchmarks do not assess provenance. SciFigPlag-Bench provides a factorized taxonomy and diagnostic tasks to evaluate multimodal models on fine-grained provenance reasoning, aiding research integrity.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0bf388a17f76af0d49accf5c30efef19253cb91f2bd019d78ea06acc39adc1e7"},"motivation":"Scientific figures often encode the visual evidence behind scientific findings, yet figure plagiarism remains underexplored as a benchmarked multimodal evaluation problem.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.29124","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_scifigqual-bench_f473d66d","familyId":"bmf_dc38c7b96fa2","name":"SciFigQual-Bench","oneLine":"A benchmark for evaluating scientific figure quality across five dimensions with full-manuscript context, using 6,308 expert-rated images from top CS conferences and a fixed eval1200 test split.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27084","pdf":"https://arxiv.org/pdf/2607.27084","project":"https://frankdengai.github.io/SciFigQual-Bench","code":"https://github.com/FrankDengAI/SciFigQual-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.27084"},"evidence":{"snippet":"To address this, we propose SciFigQual-Bench, a full-text contextual benchmark that evaluates scientific images across five dimensions (clarity, layout, caption fit, context relevance, and misleading risk).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27084"},"ranking":{"90d":{"score":15,"rank":400,"coverage":0.7,"confidence":"Medium"}},"description":"A benchmark for evaluating scientific figure quality across five dimensions with full-manuscript context, using 6,308 expert-rated images from top CS conferences and a fixed eval1200 test split.","whyItMatters":"Existing image quality assessment methods are unsuitable for scientific figures; this benchmark provides a contextual, multi-dimensional standard for automated evaluation, enabling model comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e36f909aaba8edc3ef929e6ba96e415ccd670756f5472f6d6e9e90fc4ed504a9"},"motivation":"Scientific images are the core elements of presenting experimental conclusions, elaborating system architecture, and supporting comparative arguments in scientific papers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27084","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FrankDengAI","organizationType":"academic-lab","sourceUrl":"https://github.com/FrankDengAI/SciFigQual-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_scihazard_d93c4c2a","familyId":"bmf_f9cf90b25fd3","name":"SciHazard","oneLine":"SciHazard evaluates LLMs on scientific safety risks across 12 disciplines with 2,400 hazardous and 600 oversafety questions grounded in regulated entities. The DeHarm-Score decomposes harm into executability and net-new risk, providing a detailed scoring contract.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.18665","pdf":"https://arxiv.org/pdf/2607.18665","project":"https://anonymous.4open.science/r/DeharmScore-7B55","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.18665"},"evidence":{"snippet":"To address this, we introduce SciHazard, a real-world-grounded benchmark for scientific risks and a dataset agnostic evaluation framework for measuring harmfulness.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.18665"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SciHazard evaluates LLMs on scientific safety risks across 12 disciplines with 2,400 hazardous and 600 oversafety questions grounded in regulated entities. The DeHarm-Score decomposes harm into executability and net-new risk, providing a detailed scoring contract.","whyItMatters":"Existing safety benchmarks often use templated queries and LLM-as-a-Judge without domain grounding. SciHazard offers real-world grounded evaluation, and the DeHarm-Score improves agreement with expert annotations by 90% over baselines, enabling more reliable safety measurement for scientific LLMs and agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"aefcba623125c8927f039fa89c0054a27f726f3bc8f10a4ff70e94fb3210c20e"},"motivation":"Large language models (LLMs) increasingly support science, but they can also convert hazardous scientific knowledge into actionable misuse guidance.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.18665","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DeharmScore GitHub","organizationType":"community","sourceUrl":"https://anonymous.4open.science/r/DeharmScore-7B55","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_sciintbench_70ffaf60","familyId":"bmf_c2dee4bdf202","name":"SciIntBench","oneLine":"SciIntBench is an adversarial benchmark of 810 prompts across ten responsible-conduct-of-research categories and three scientific domains. Each scenario appears in overt, covert, and benign versions to measure framing-sensitive refusal of misconduct.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.29468","pdf":"https://arxiv.org/pdf/2605.29468","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29468"},"evidence":{"snippet":"We introduce SciIntBench, an adversarial benchmark of 810 prompts across ten RCR categories and three scientific domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29468"},"ranking":{},"description":"SciIntBench is an adversarial benchmark of 810 prompts across ten responsible-conduct-of-research categories and three scientific domains. Each scenario appears in overt, covert, and benign versions to measure framing-sensitive refusal of misconduct.","whyItMatters":"LLMs are increasingly used in scientific work, but their compliance with research integrity norms is unclear. SciIntBench aims to measure how models handle framing-sensitive ethical scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9f043f3375d17a44f4435fdea0fb7a6dd8fa673dff8f1aff58d33b75d043e2a9"},"motivation":"Large language models (LLMs) are increasingly used to support scientific work, but it is unclear whether they uphold responsible conduct of research (RCR) norms or help undermine them.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29468","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_sciir-bench_0a7d1ae8","familyId":"bmf_e6067af2051a","name":"SciIR-Bench","oneLine":"SciIR-Bench evaluates text-to-image models on scientific image reasoning across three semiotic-aligned tracks: entity structure, scientific process, and scientific law. It uses an atomic checklist to convert scientific accuracy into verifiable fine-grained questions.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.30124","pdf":"https://arxiv.org/pdf/2606.30124","project":null,"code":"https://github.com/MAIR-Lab-HUST/SciIR","data":null,"hfPaper":"https://huggingface.co/papers/2606.30124"},"evidence":{"snippet":"For evaluation, we propose SciIR-Bench, which aligns with these three semiotic levels and employs an Atomic Checklist to convert the outcome-oriented scientific accuracy into process-oriented, verifiable, fine-grained questions.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":"2026-07-02T00:00:00.000Z","githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.30124"},"ranking":{"90d":{"score":35,"rank":193,"coverage":0.7,"confidence":"Medium"}},"description":"SciIR-Bench evaluates text-to-image models on scientific image reasoning across three semiotic-aligned tracks: entity structure, scientific process, and scientific law. It uses an atomic checklist to convert scientific accuracy into verifiable fine-grained questions.","whyItMatters":"Current text-to-image models lack rigorous evaluation for scientific imagery, which requires logical reasoning beyond visual fidelity. SciIR-Bench provides a structured protocol to measure such capabilities, aiding selection and development of models for scientific visualization tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"83d731dbb3f312a2f5e2092f2b709ab7f0034d8f9dd4ec634cb4c2f8520ef193"},"motivation":"While Text-to-Image (T2I) models have shown remarkable success in generating photorealistic visual content, they still struggle with the rigorous semantic alignment and logical reasoning required for scientific imagery.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026","evidence":"Accepted to ECCV 2026","evidenceUrl":"https://arxiv.org/abs/2606.30124","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026","reviewStatus":"accepted","decisionRaw":"Accepted to ECCV 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.30124","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to ECCV 2026","level":"author-claim"}]}],"publishers":[{"name":"MAIR-Lab-HUST","organizationType":"academic-lab","sourceUrl":"https://github.com/MAIR-Lab-HUST/SciIR","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_scimif_1f2ceaa4","familyId":"bmf_ccaafb36d317","name":"SciMIF","oneLine":"Evaluates instruction following of MLLMs across five scientific disciplines with a taxonomy of 10 constraint groups.","area":"Multimodal","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-26","firstSeenAt":"2026-08-27","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25973","pdf":"https://arxiv.org/pdf/2608.25973","project":null,"code":"https://github.com/shenye7436/SciMIF","data":null,"hfPaper":null},"evidence":{"snippet":"In this work, we introduce SciMIF, a novel benchmark designed to evaluate the capability of MLLMs in following complex scientific instructions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25973"},"ranking":{"30d":{"score":23,"rank":109,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":313,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates instruction following of MLLMs across five scientific disciplines with a taxonomy of 10 constraint groups.","whyItMatters":"Fills the gap in multimodal instruction adherence in scientific applications, revealing discipline-specific challenges.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"146f4921e18f6b3ba7d73a92a72aa0cb8365b7652b30d98cb7a01c3a7f3dd0de"},"motivation":"Understanding instruction-following capabilities in scientific domains is essential for effectively leveraging Multimodal Large Language Models (MLLMs) to advance the development of scientific fields.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:12:36.511123Z","model":"deepseek-v4-pro","decisionReason":"The code repository provides a construction pipeline and evaluation scripts, enabling replication and extension.","canonicalNameSource":"paper_title","canonicalNameEvidence":"SciMIF: Understanding Multimodal Instruction Following in Scientific Domains"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25973","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":15,"confidence":"Low","horizon":"7d","reason":"Scientific multimodal instruction following is a niche but emerging area; public code may drive modest interest."},"evaluationMode":"public_reusable","publishers":[{"name":"Shen Ye","organizationType":"academic-lab","sourceUrl":"https://github.com/shenye7436/SciMIF","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_scir_5305af40","familyId":"bmf_fdd9b59047b7","name":"SciR","oneLine":"SciR evaluates LLMs on deduction, induction, and causal abduction in scientific settings, with tasks generated from formal objects and rendered into multi-document scientific discourse. Difficulty is controlled along extraction and inference axes, with verifiable answers.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13020","pdf":"https://arxiv.org/pdf/2606.13020","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.13020"},"evidence":{"snippet":"We introduce SciR, a benchmark that combines multi-paradigm reasoning with controllable scientific rendering, anchored on three paradigmatic scientific problems.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13020"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SciR evaluates LLMs on deduction, induction, and causal abduction in scientific settings, with tasks generated from formal objects and rendered into multi-document scientific discourse. Difficulty is controlled along extraction and inference axes, with verifiable answers.","whyItMatters":"Existing benchmarks either lack mechanistic ground truth or do not resemble real scientific documents. SciR provides a controllable protocol for isolating extraction vs. inference failures, which is valuable for diagnosing model capabilities in scientific reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3e4c1dad6af2a8ed0bc0ad7f976f06ef9e4ea462214a544d574a00883e5f946d"},"motivation":"Three paradigmatic forms of inference recur across scientific reasoning: deduction, induction, and causal abduction.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13020","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_scirisk-bench_01fe7487","familyId":"bmf_700b9be2aca9","name":"SciRisk-Bench","oneLine":"SciRisk-Bench evaluates AI4Science safety across 7 disciplines and 10 risk dimensions, assessing whether models recognize and avoid risks in scientific contexts.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18936","pdf":"https://arxiv.org/pdf/2606.18936","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18936"},"evidence":{"snippet":"We introduce \\textbf{SciRisk-Bench}, a benchmark designed to evaluate AI4Science safety from two complementary perspectives: explicit risk dimensions and scientific disciplines.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18936"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SciRisk-Bench evaluates AI4Science safety across 7 disciplines and 10 risk dimensions, assessing whether models recognize and avoid risks in scientific contexts.","whyItMatters":"Existing AI4Science safety benchmarks lack explicit risk dimensions; SciRisk-Bench provides fine-grained diagnosis of safety issues across disciplines.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"69d4859f097c9b756f5ed7ded2d6ddc8fe2d2f3ba138678da77cb8fa4cf7da45"},"motivation":"Large language models (LLMs) are increasingly embedded in AI for Science (AI4Science) workflows, from scientific question answering and literature analysis to laboratory planning and autonomous discovery.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18936","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_scistylebench_232be1cc","familyId":"bmf_087086a27a87","name":"SciStyleBench","oneLine":"SciStyleBench diagnoses stylistic bias in LLM-based idea evaluation through controlled stylistic perturbations, metrics, and a mitigation extractor.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.01666","pdf":"https://arxiv.org/pdf/2608.01666","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01666"},"evidence":{"snippet":"To address this question, we propose SciStyleBench, a unified three-component benchmark for diagnosing and mitigating stylistic bias in LLM-based idea evaluation: (i) First, SciStyleStage, a three-stage evaluation environment that applies controlled stylistic perturbations to fixed scientific content across three settings no context, fixed-domain context, and open-domain retrieval context, covering 600 scientific ideas and 15 style variants, with 9,000 evaluation instances per setting; (ii) Seco","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01666"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SciStyleBench diagnoses stylistic bias in LLM-based idea evaluation through controlled stylistic perturbations, metrics, and a mitigation extractor.","whyItMatters":"Highlights the importance of style invariance in scientific idea evaluation and provides metrics and a mitigation module for improving LLM judges.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c5fa9ae335618287682507905754382d6f7077a6f33702e05d89bf2c1e48151a"},"motivation":"However, whether these judges truly evaluate the scientific substance of ideas or are influenced by superficial stylistic presentation remains an open question.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.01666","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_220724a844e90df3","familyId":"catalog_family_220724a844e90df3","name":"SCONE post-cutoff success","oneLine":"Share of SCONE smart-contract vulnerabilities exploited on the 12-task post-cutoff set.","description":"Share of SCONE smart-contract vulnerabilities exploited on the 12-task post-cutoff set.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.anthropic.com/research/exploit-evals","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_220724a844e90df3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/sconepostcutoffsuccess"}],"catalogSources":[{"catalog":"benchlm","sourceId":"sconePostCutoffSuccess","url":"https://benchlm.ai/benchmarks/sconepostcutoffsuccess","paperUrl":"https://www.anthropic.com/research/exploit-evals","year":"2026","fullName":"SCONE Post-Cutoff Exploit Success","format":"Best@8 exploit success rate","tasks":"12 post-cutoff smart-contract vulnerabilities","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d276c9c4bbd0eb85","familyId":"catalog_family_d276c9c4bbd0eb85","name":"SCONE simulated revenue","oneLine":"Simulated value captured across the SCONE post-cutoff smart-contract set.","description":"Simulated value captured across the SCONE post-cutoff smart-contract set.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.anthropic.com/research/exploit-evals","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d276c9c4bbd0eb85"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/sconepostcutoffrevenueusdm"}],"catalogSources":[{"catalog":"benchlm","sourceId":"sconePostCutoffRevenueUsdM","url":"https://benchlm.ai/benchmarks/sconepostcutoffrevenueusdm","paperUrl":"https://www.anthropic.com/research/exploit-evals","year":"2026","fullName":"SCONE Post-Cutoff Simulated Exploit Revenue","format":"Simulated USD millions, Best@8","tasks":"12 post-cutoff smart-contract vulnerabilities","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_scope-bench_aaf50d3b","familyId":"bmf_7f2b8edd4cc8","name":"SCOPE-Bench","oneLine":"SCOPE-Bench evaluates short-video recommendation systems by quantifying content depth using the Content Depth Score (CDS), a seven-level scale based on cognitive psychology. It provides CDS annotations for 150K videos from an open-source dataset, enabling systematic assessment of recommenders from a cognitive-content perspective.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.13990","pdf":"https://arxiv.org/pdf/2608.13990","project":"https://liweidengdavid.github.io/SCOPE-Bench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.13990"},"evidence":{"snippet":"As an initial step toward this vision, we present \\textbf{SCOPE-Bench}, the first benchmark for content-depth evaluation in short-video recommendation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13990"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SCOPE-Bench evaluates short-video recommendation systems by quantifying content depth using the Content Depth Score (CDS), a seven-level scale based on cognitive psychology. It provides CDS annotations for 150K videos from an open-source dataset, enabling systematic assessment of recommenders from a cognitive-content perspective.","whyItMatters":"Existing short-video recommenders optimize for engagement, often favoring shallow content. SCOPE-Bench addresses the lack of benchmarks for content depth in recommendation, allowing evaluation of algorithms on their ability to recommend cognitively deep content, which has implications for user well-being and long-term engagement.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d9196b6b479a5eaaf7bffcd39868056996e167973770566ffa342a1e1b10834f"},"motivation":"Driven by the attention economy, short-video Recommender Systems (RSs) are primarily optimized to maximize user engagement by promoting videos that capture attention within seconds.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13990","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_scr-bench_984f0fe8","familyId":"bmf_ae50c3042f6a","name":"SCR-Bench","oneLine":"SCR-Bench is a benchmark for evaluating security risks in composed LLM agent skill workflows. It includes three sub-benchmarks (SCR-CapFlow, SCR-TrustLift, SCR-AuthBlur) that measure attack success rates, trust transfer, and authorization confusion in sandboxed environments.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15242","pdf":"https://arxiv.org/pdf/2606.15242","project":null,"code":"https://github.com/saint-viperx/SCR_Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.15242"},"evidence":{"snippet":"We introduce SCR-Bench to evaluate this risk in controlled, sandboxed skill environments.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":14,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15242"},"ranking":{"90d":{"score":37,"rank":171,"coverage":0.7,"confidence":"Medium"}},"description":"SCR-Bench is a benchmark for evaluating security risks in composed LLM agent skill workflows. It includes three sub-benchmarks (SCR-CapFlow, SCR-TrustLift, SCR-AuthBlur) that measure attack success rates, trust transfer, and authorization confusion in sandboxed environments.","whyItMatters":"SCR-Bench addresses the gap in evaluating agent skill security at the path level, where skills benign in isolation become harmful in composition. It provides a controlled, sandboxed environment for assessing composition-induced risks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"aa29672772df59586e700ba9db9bc6f2001ea20ee085d0daea2c6c162a3a8f47"},"motivation":"Skills are becoming the capability layer through which LLM agents turn plans into actions, but their use introduces security risks such as data leakage, unauthorized operations, and tool misuse.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15242","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SCR-Bench Team","organizationType":"academic-lab","sourceUrl":"https://github.com/saint-viperx/SCR_Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_scrambletoolbench_aa15193b","familyId":"bmf_2758dc7261ac","name":"ScrambleToolBench","oneLine":"ScrambleToolBench evaluates agent behavioral reasoning in an interactive terminal environment with obfuscated tools and dynamic challenges like mapping drift and stochastic failures.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02358","pdf":"https://arxiv.org/pdf/2608.02358","project":null,"code":"https://github.com/declare-lab/ScrambleToolBench","data":null,"hfPaper":"https://huggingface.co/papers/2608.02358"},"evidence":{"snippet":"To address this limitation, we introduce ScrambleToolBench, an interactive terminal benchmark designed to isolate behavioral reasoning.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":11,"hfDailySubmittedAt":"2026-08-04T00:00:00.000Z","githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02358"},"ranking":{"30d":{"score":34,"rank":63,"coverage":0.85,"confidence":"High"},"90d":{"score":34,"rank":205,"coverage":0.7,"confidence":"Medium"}},"description":"ScrambleToolBench evaluates agent behavioral reasoning in an interactive terminal environment with obfuscated tools and dynamic challenges like mapping drift and stochastic failures.","whyItMatters":"Isolates behavioral reasoning from prior knowledge, revealing gaps in agent adaptation and supporting comparison of agent architectures.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3e01134c43707be0ccc50816693ac98caa4b3846a76af61e086562584d3efb0f"},"motivation":"To operate robustly in open-world environments, autonomous agents should be able to infer the behavior of unfamiliar systems through interaction alone, even in the absence of documentation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02358","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_scratchworld_04c08f86","familyId":"bmf_0395c14a6789","name":"ScratchWorld","oneLine":"ScratchWorld is an offline diagnostic benchmark for world models, using Scratch projects as executable worlds. It evaluates next-state prediction, long-horizon tracking, causal attribution, and counterfactual prediction with replay-verified transitions.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.31689","pdf":"https://arxiv.org/pdf/2606.31689","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.31689"},"evidence":{"snippet":"We introduce ScratchWorld, an offline diagnostic benchmark that treats Scratch projects as executable worlds and uses a pinned Scratch VM to produce replay-verified transitions, hidden variables, causal traces, and counterfactual outcomes.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31689"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ScratchWorld is an offline diagnostic benchmark for world models, using Scratch projects as executable worlds. It evaluates next-state prediction, long-horizon tracking, causal attribution, and counterfactual prediction with replay-verified transitions.","whyItMatters":"Provides a new evaluation paradigm for world models that avoids confounds like overlap with persistent state. Enables direct comparison of model capabilities on executable consequences.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"20ec2b91a1225ff4419b7bcc3e7d4df8f4a8ece6b5522b0356cb2d3390f43c93"},"motivation":"World-model evaluations often score a predicted future by overlap with a target state or observation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31689","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_70f048edc414a304","familyId":"catalog_family_70f048edc414a304","name":"ScreenSpot","oneLine":"ScreenSpot is the first realistic GUI grounding benchmark that encompasses mobile, desktop, and web environments. The dataset comprises over 1,200 instructions from iOS, Android, macOS, Windows and Web environments, along with annotated element types (text and icon/widget), designed to evaluate visual GUI agents' ability to accurately locate screen elements based on natural language instructions.","description":"ScreenSpot is the first realistic GUI grounding benchmark that encompasses mobile, desktop, and web environments. The dataset comprises over 1,200 instructions from iOS, Android, macOS, Windows and Web environments, along with annotated element types (text and icon/widget), designed to evaluate visual GUI agents' ability to accurately locate screen elements based on natural language instructions.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Spatial Reasoning","Grounding","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/screenspot","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_70f048edc414a304"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/screenspot"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"screenspot","url":"https://llm-stats.com/benchmarks/screenspot","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","spatial reasoning","grounding","vision"],"catalogModelCount":16,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_8b6ef0a34d9d5d68","familyId":"catalog_family_8b6ef0a34d9d5d68","name":"ScreenSpot Pro","oneLine":"ScreenSpot-Pro is a novel GUI grounding benchmark designed to rigorously evaluate the grounding capabilities of multimodal large language models (MLLMs) in professional high-resolution computing environments. The benchmark comprises 1,581 instructions across 23 applications spanning 5 industries and 3 operating systems, featuring authentic high-resolution images from professional domains with expert annotations. Unlike previous benchmarks that focus on cropped screenshots in consumer applications, ScreenSpot-Pro addresses the complexity and diversity of real-world professional software scenarios, revealing significant performance gaps in current MLLM GUI perception capabilities.","description":"ScreenSpot-Pro is a novel GUI grounding benchmark designed to rigorously evaluate the grounding capabilities of multimodal large language models (MLLMs) in professional high-resolution computing environments. The benchmark comprises 1,581 instructions across 23 applications spanning 5 industries and 3 operating systems, featuring authentic high-resolution images from professional domains with expert annotations. Unlike previous benchmarks that focus on cropped screenshots in consumer applications, ScreenSpot-Pro addresses the complexity and diversity of real-world professional software scenarios, revealing significant performance gaps in current MLLM GUI perception capabilities.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Spatial Reasoning","Grounding","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2504.07981","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8b6ef0a34d9d5d68"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/screenspot-pro"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/screenspot-pro"}],"catalogSources":[{"catalog":"benchlm","sourceId":"screenSpotPro","url":"https://benchlm.ai/benchmarks/screenspot-pro","paperUrl":"https://arxiv.org/abs/2504.07981","year":"2025","fullName":"ScreenSpot Pro","format":"Static interface element localization","tasks":"1,581 grounding instructions","successorKey":null},{"catalog":"llm-stats","sourceId":"screenspot-pro","url":"https://llm-stats.com/benchmarks/screenspot-pro","datasetSlug":"screenspot-pro","versionCount":1,"subsetCount":27,"rowCount":null,"updatedAt":"2026-05-29T23:11:08.897266+00:00","community":true}],"catalogCategories":["multimodalGrounded","multimodal","spatial reasoning","grounding","vision"],"catalogModelCount":25,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_sctranslation_c3e5eec8","familyId":"bmf_9b7fdc707061","name":"scTranslation","oneLine":"scTranslation evaluates single-cell multi-omics modality translation across RNA, ATAC, and ADT modalities, with diverse datasets, six representative models, and comprehensive metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03906","pdf":"https://arxiv.org/pdf/2606.03906","project":null,"code":"https://github.com/Bunnybeibei/scTranslation","data":null,"hfPaper":"https://huggingface.co/papers/2606.03906"},"evidence":{"snippet":"To address this, we present scTranslation, a comprehensive benchmark for single-cell multi-omics modality translation tasks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":8,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03906"},"ranking":{"90d":{"score":39,"rank":161,"coverage":0.55,"confidence":"Low"}},"description":"scTranslation evaluates single-cell multi-omics modality translation across RNA, ATAC, and ADT modalities, with diverse datasets, six representative models, and comprehensive metrics.","whyItMatters":"Addresses the lack of systematic benchmarks for single-cell modality translation, offering standardized datasets, metrics, and model integration to support reproducible comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"94250e1e471f7426ac5c0e7e70431cd12042a37e13e8ed4a365ac238426fde9e"},"motivation":"Simultaneous measurement of multiple omics modalities in single cells enables researchers to gain a more comprehensive understanding of cellular states and regulatory mechanisms.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03906","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"scTranslation team","organizationType":"community","sourceUrl":"https://github.com/Bunnybeibei/scTranslation","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_sdabench_9285d4fe","familyId":"bmf_4b6f310c64ce","name":"SDABench","oneLine":"SDABench evaluates LLMs' scientific data analysis capabilities across six capabilities (descriptive, exploratory, inferential, predictive, causal, mechanistic) and five domains, with 527 real and 6000 synthetic instances in multiple-choice and open-ended formats.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals"],"capabilities":[],"topics":["AI Scientist"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.11079","pdf":"https://arxiv.org/pdf/2607.11079","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.11079"},"evidence":{"snippet":"We introduce SDABench, a benchmark that reorganizes evaluation around six capabilities (descriptive, exploratory, inferential, predictive, causal, and mechanistic) across five domains (Biology, Chemistry, Environment, Geography, Physics).","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":10,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.11079"},"ranking":{"90d":{"score":49,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SDABench evaluates LLMs' scientific data analysis capabilities across six capabilities (descriptive, exploratory, inferential, predictive, causal, mechanistic) and five domains, with 527 real and 6000 synthetic instances in multiple-choice and open-ended formats.","whyItMatters":"This benchmark reveals that LLMs degrade sharply on tasks requiring assumption selection, latent-process modeling, and mechanistic reasoning, highlighting gaps for scientific discovery applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"07fdbbabecc9cf01fcdb9fc894ad3d1ad4a37be1b3a46ad553feef967c6d090b"},"motivation":"Existing benchmarks for scientific data analysis evaluate LLMs primarily on code execution or workflow completion, overlooking that scientific analysis serves to support distinct types of scientific claims: hypothesis exploration, statistical inference, mechanistic explanation, each with different assumptions and validity criteria.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.11079","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_sdgbiasbench_aa7777ae","familyId":"bmf_7819f7afec07","name":"SDGBiasBench","oneLine":"SDGBiasBench is a benchmark suite for evaluating vision-language models on Sustainable Development Goals (SDG) reasoning, covering 500k multiple-choice questions and 50k regression tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.21919","pdf":"https://arxiv.org/pdf/2605.21919","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.21919"},"evidence":{"snippet":"To address this gap, we propose SDGBiasBench, a large-scale benchmark suite for SDG-oriented vision-language reasoning.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.21919"},"ranking":{},"description":"SDGBiasBench is a benchmark suite for evaluating vision-language models on Sustainable Development Goals (SDG) reasoning, covering 500k multiple-choice questions and 50k regression tasks.","whyItMatters":"Existing SDG evaluation tools lack a combined assessment of qualitative and quantitative reasoning, and this benchmark aims to expose systematic biases in model predictions. The value lies in enabling more reliable AI for sustainable development monitoring.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6dc257974a66feb822577a03a7b8377bdfba5ddf19c19f1218ec7b61d143cf99"},"motivation":"Assessing progress toward the Sustainable Development Goals (SDGs) requires multi-step reasoning over visual cues, contextual knowledge, and development indicators, where incomplete evidence use and imperfect evidence integration can introduce hidden prediction biases.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 Findings","evidence":"Accepted to EMNLP 2026 Findings","evidenceUrl":"https://arxiv.org/abs/2605.21919","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"EMNLP 2026 Findings","reviewStatus":"accepted","decisionRaw":"Accepted to EMNLP 2026 Findings","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2605.21919","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to EMNLP 2026 Findings","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_sdr-bench_b95d0afe","familyId":"bmf_78c11886d9c2","name":"SDR-Bench","oneLine":"SDR-Bench evaluates LLM personalization in a two-party framework. It includes 6,279 customer success stories across 22 industries and ~200 enterprises, with a temporally constrained simulation to prevent data leakage. Scoring compares model-generated outreach against human outcomes.","area":"Language & Knowledge","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Logistics"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.20471","pdf":"https://arxiv.org/pdf/2607.20471","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20471"},"evidence":{"snippet":"We release SDR-Bench, a public corpus of 6,279 customer success stories spanning 22 industries and approximately 200 enterprises, served through a temporally constrained simulation that prevents future-data leakage.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20471"},"ranking":{},"description":"SDR-Bench evaluates LLM personalization in a two-party framework. It includes 6,279 customer success stories across 22 industries and ~200 enterprises, with a temporally constrained simulation to prevent data leakage. Scoring compares model-generated outreach against human outcomes.","whyItMatters":"SDR-Bench addresses the gap in evaluating LLM personalization as a two-party problem, where generated messages must induce action in a third party. It provides a reproducible public benchmark for comparing model performance on this task, with validation against field deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c6b19bbe38a867ea12575e93ca406cc3e43b9e476b627ffac8d446f4a6baa118"},"motivation":"Personalization, the act of varying a message to induce action from a specific receiver while keeping sender, channel, and time fixed, has a long tradition in psychology and marketing as a two-party problem in which sender and receiver have independent objectives.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20471","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_573169b4bacca2e3","familyId":"catalog_family_573169b4bacca2e3","name":"Seal-0","oneLine":"Seal-0 is a benchmark for evaluating agentic search capabilities, testing models' ability to navigate and retrieve information using tools.","description":"Seal-0 is a benchmark for evaluating agentic search capabilities, testing models' ability to navigate and retrieve information using tools.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Search"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/seal-0","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_573169b4bacca2e3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/seal-0"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"seal-0","url":"https://llm-stats.com/benchmarks/seal-0","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","search"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_seam-bench_df6f1e82","familyId":"bmf_13bb3b50c414","name":"SEAM-Bench","oneLine":"SEAM-Bench is a double-blind continuity storyboarding benchmark for evaluating visual continuity in short-drama generation. It assesses cross-episode continuity recall and generalizes across six mainstream text models.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-26","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2608.22725","pdf":"https://arxiv.org/pdf/2608.22725","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We further release SEAM-Bench, a double-blind continuity storyboarding benchmark, on which SEAM raises cross-episode continuity recall from 0.700 to 0.946, generalizes across six mainstream text models, and yields consistent, though not yet significant, gains at the generated-image layer.","reasonCodes":["exact named benchmark artifact released in abstract"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22725"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SEAM-Bench is a double-blind continuity storyboarding benchmark for evaluating visual continuity in short-drama generation. It assesses cross-episode continuity recall and generalizes across six mainstream text models.","whyItMatters":"The benchmark fills the gap in evaluating visual continuity in large-scale short-drama generation, which is critical for production pipelines. It provides a standardized protocol to compare memory-based approaches and guide improvements in episodic generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"6b4efb7d483e2a5eca0d8e902962ab8b9392221040573ca9055b12767e27249f"},"motivation":"Short-drama generation has grown into a large, industrialized pipeline, and as it scales from isolated shots to the episode level, visual continuity has become a critical bottleneck.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally released with a double-blind evaluation setup and is used to compare SEAM against baselines, meeting the criteria for a published benchmark with a stable scoring contract.","canonicalNameSource":"abstract","canonicalNameEvidence":"We further release SEAM-Bench, a double-blind continuity storyboarding benchmark"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.22725","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":35,"confidence":"Low","horizon":"7d","reason":"The benchmark is specific to short-drama generation with moderate interest, but its narrow scope may limit initial attention."},"evaluationMode":"score_submission","publishers":[{"name":"CreativeFitting","organizationType":"company-research-lab","sourceUrl":"https://arxiv.org/abs/2608.22725","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_c083b66e11dca90a","familyId":"catalog_family_c083b66e11dca90a","name":"Search and Function-Calling","oneLine":"Search and Function-Calling is an OpenAI internal production benchmark measuring reliable search-tool use and function calling in agentic workflows, reported as a pass rate.","description":"Search and Function-Calling is an OpenAI internal production benchmark measuring reliable search-tool use and function calling in agentic workflows, reported as a pass rate.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/openai-search-function-calling","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c083b66e11dca90a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/openai-search-function-calling"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"openai-search-function-calling","url":"https://llm-stats.com/benchmarks/openai-search-function-calling","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","tool calling"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_sec-bench_29fdf2d8","familyId":"bmf_830a56b48245","name":"SEC-bench","oneLine":"SEC-bench Pro measures long-horizon vulnerability discovery in software security. It includes 344 verified vulnerabilities across V8, SpiderMonkey, and Linux kernel, each paired with instructions for reproducing a working proof-of-concept. Grading uses an LLM-based judge to classify generated PoCs against vulnerable, fixed, and latest images.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.26548","pdf":"https://arxiv.org/pdf/2605.26548","project":null,"code":"https://github.com/SEC-bench/SEC-bench-Pro","data":null,"hfPaper":"https://huggingface.co/papers/2605.26548"},"evidence":{"snippet":"We present SEC-bench Pro, a benchmark that measures how well frontier models hunt real vulnerabilities by reproducing working PoC inputs from disclosed reports, where each task pairs a concrete bug with the instructions for triggering it.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":47,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26548"},"ranking":{},"description":"SEC-bench Pro measures long-horizon vulnerability discovery in software security. It includes 344 verified vulnerabilities across V8, SpiderMonkey, and Linux kernel, each paired with instructions for reproducing a working proof-of-concept. Grading uses an LLM-based judge to classify generated PoCs against vulnerable, fixed, and latest images.","whyItMatters":"Finding real vulnerabilities requires reasoning across an entire codebase to produce a working PoC, a challenging task that is understudied. SEC-bench Pro provides a reproducible environment for evaluating long-horizon security capabilities, helping identify where models succeed and fail in realistic vulnerability hunting.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4a67b69f1e510b57e99bbf96d2c727261215a3f297a7c61b91228934e693307a"},"motivation":"Finding a real vulnerability in complicated systems is a challenging, long-horizon task that demands reasoning across an entire codebase to produce a working proof-of-concept (PoC).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.26548","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_357a19bc43aba891","familyId":"catalog_family_357a19bc43aba891","name":"SEC-Bench Pro","oneLine":"SEC-bench Pro is a self-evolving software-security benchmark that measures agent bug hunting on critical, high-complexity systems. It instantiates validated vulnerabilities across the V8 and SpiderMonkey JavaScript engines as reproducible vulnerability-discovery and proof-of-concept-generation tasks with oracle-based validation.","description":"SEC-bench Pro is a self-evolving software-security benchmark that measures agent bug hunting on critical, high-complexity systems. It instantiates validated vulnerabilities across the V8 and SpiderMonkey JavaScript engines as reproducible vulnerability-discovery and proof-of-concept-generation tasks with oracle-based validation.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External","Safety","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/gpt-5-6/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_357a19bc43aba891"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/secbenchpro"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/sec-bench-pro"}],"catalogSources":[{"catalog":"benchlm","sourceId":"secBenchPro","url":"https://benchlm.ai/benchmarks/secbenchpro","paperUrl":"https://openai.com/index/gpt-5-6/","year":"2026","fullName":"SEC-Bench Pro","format":"Success rate","tasks":"Security engineering tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"sec-bench-pro","url":"https://llm-stats.com/benchmarks/sec-bench-pro","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["external","safety","agents","code"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_68f604f200b08b50","familyId":"catalog_family_68f604f200b08b50","name":"SecCodeBench","oneLine":"SecCodeBench evaluates LLM coding agents on secure code generation and vulnerability detection, testing the ability to produce code that is both functional and free from security vulnerabilities.","description":"SecCodeBench evaluates LLM coding agents on secure code generation and vulnerability detection, testing the ability to produce code that is both functional and free from security vulnerabilities.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/seccodebench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_68f604f200b08b50"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/seccodebench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"seccodebench","url":"https://llm-stats.com/benchmarks/seccodebench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_secdrift_836e6631","familyId":"bmf_10522f747d5f","name":"SecDrift","oneLine":"SecDrift is a framework measuring sector-conditioned security drift in AI-generated code, evaluating vulnerability rates from LLMs when prompted with industry contexts versus neutral baselines across CISA sectors and CWE categories.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25225","pdf":"https://arxiv.org/pdf/2607.25225","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.25225"},"evidence":{"snippet":"We present SecDrift, a benchmark measuring sector-conditioned security drift: the change in static-analysis vulnerability rates when prompts are conditioned on industry contexts versus neutral baselines.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25225"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SecDrift is a framework measuring sector-conditioned security drift in AI-generated code, evaluating vulnerability rates from LLMs when prompted with industry contexts versus neutral baselines across CISA sectors and CWE categories.","whyItMatters":"It addresses whether domain-specific prompting affects code security, with findings that model choice matters more than prompt framing. The framework could guide deployment decisions, but its standalone benchmark status is unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ac53e40e73e6cff85237b5f26815fd1b3dd70242a7655bfd65c06c2981f802a3"},"motivation":"LLMs are increasingly used for code generation in critical infrastructure, yet the security effect of domain-specific prompting is understudied.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25225","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_secrespond_a48429f2","familyId":"bmf_9705c5bae016","name":"SecRespond","oneLine":"A benchmark for evaluating LLM agents on post-compromise incident-response workflows using forensic disk snapshots and host security reports across 10 cyber ranges.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.26791","pdf":"https://arxiv.org/pdf/2607.26791","project":null,"code":"https://github.com/Alibaba-NLP/qqr/tree/main/data/secrespond","data":null,"hfPaper":"https://huggingface.co/papers/2607.26791"},"evidence":{"snippet":"To address this gap, we introduce SecRespond, the first benchmark for evaluating LLM agents on the post-compromise incident-response workflow.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":4,"hfDailySubmittedAt":"2026-07-30T00:00:00.000Z","githubStars":283,"githubScope":"hosting_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26791"},"ranking":{"90d":{"score":47,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"A benchmark for evaluating LLM agents on post-compromise incident-response workflows using forensic disk snapshots and host security reports across 10 cyber ranges.","whyItMatters":"Cybersecurity benchmarks typically focus on pre-compromise settings; this addresses the gap in evaluating agents for real-world incident response, where proactive investigation and remediation are critical.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6f9ef85dd26044adf9cfbb4910f7395137fb5d60b1f2112ec1964a99e2556f2f"},"motivation":"Large Language Model (LLM) agents are increasingly adopted in real-world security operations with access to host artifacts and command-line interfaces (CLIs), making it critical to thoroughly assess their security capabilities.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26791","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Alibaba NLP","organizationType":"company-research-lab","sourceUrl":"https://github.com/Alibaba-NLP/qqr/tree/main/data/secrespond","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_8a02548a7abf79a1","familyId":"catalog_family_8a02548a7abf79a1","name":"SeedClawBench","oneLine":"SeedClawBench is an agentic coding benchmark measuring overall model performance on real-world, tool-using software development tasks.","description":"SeedClawBench is an agentic coding benchmark measuring overall model performance on real-world, tool-using software development tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agents","Code","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/seedclawbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8a02548a7abf79a1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/seedclawbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"seedclawbench","url":"https://llm-stats.com/benchmarks/seedclawbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code","tool calling"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_segabench_e418e029","familyId":"bmf_7c6281aa1c49","name":"SeGaBench","oneLine":"SeGaBench is an executable benchmark containing 120 cases (100 synthetic, 20 source-backed) to test whether LLMs can recover semantic optimization opportunities that compilers miss, with hidden enabling semantics, oracle artifacts, and validators.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.PL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.03983","pdf":"https://arxiv.org/pdf/2608.03983","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03983"},"evidence":{"snippet":"We introduce SeGaBench, an executable benchmark containing 100 synthetic and 20 source-backed cases spanning low-level assumptions, data-structure invariants, and high-level semantic lifting.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03983"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SeGaBench is an executable benchmark containing 120 cases (100 synthetic, 20 source-backed) to test whether LLMs can recover semantic optimization opportunities that compilers miss, with hidden enabling semantics, oracle artifacts, and validators.","whyItMatters":"Compilers miss profitable transformations when enabling semantics are absent from the program representation. SeGaBench evaluates whether LLMs can recover such semantics and produce validated, performance-improving artifacts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d2b994507120a294e65116dbc46b3ccb8ae5d84c4912a6cf3e33f516c33fe3df"},"motivation":"Optimizing compilers miss profitable transformations when their enabling semantics are absent from the analyzed program representation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03983","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_selectbench_07886b3c","familyId":"bmf_4efc6a551cf2","name":"SelectBench","oneLine":"SelectBench evaluates selective evidence adoption in retrieval-augmented language models, focusing on rejecting deceptive content. The benchmark includes a 325-example test set and rule- or judge-based reward scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20090","pdf":"https://arxiv.org/pdf/2607.20090","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20090"},"evidence":{"snippet":"We introduce SelectBench, a controlled benchmark and training set for selective evidence adoption, and post-train Qwen3.5-4B directly with DAPO using either deterministic rule rewards or a frozen semantic judge.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20090"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SelectBench evaluates selective evidence adoption in retrieval-augmented language models, focusing on rejecting deceptive content. The benchmark includes a 325-example test set and rule- or judge-based reward scoring.","whyItMatters":"Retrieval-augmented models often face mixed contexts with misleading content. A standardized evaluation for selective evidence adoption helps measure safety and reliability in real-world deployments.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d1030938d7ac42af52becfd0cf3cb6f9321d83525ae43d464c3715c1482d7197"},"motivation":"Retrieval-augmented large language models frequently face contexts that interleave useful evidence with misleading statements or instruction-like content.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20090","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_7a49990212bb093a","familyId":"catalog_family_7a49990212bb093a","name":"Senior SWE-Bench","oneLine":"A Snorkel AI benchmark of senior-level software engineering tasks emphasizing under-specified feature work, bug/performance investigation, and taste-based correctness.","description":"A Snorkel AI benchmark of senior-level software engineering tasks emphasizing under-specified feature work, bug/performance investigation, and taste-based correctness.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://senior-swe-bench.snorkel.ai/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7a49990212bb093a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/seniorswebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"seniorSweBench","url":"https://benchlm.ai/benchmarks/seniorswebench","paperUrl":"https://senior-swe-bench.snorkel.ai/","year":"2026","fullName":"Senior SWE-Bench","format":"Agentic software-engineering evaluation","tasks":"Senior-level repository tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_sense-vad_052128b1","familyId":"bmf_a69f1f20096d","name":"SENSE-VAD","oneLine":"SENSE-VAD is a synthetic video anomaly detection dataset for autonomous driving, generated with CARLA and Unreal Engine. It includes socially complex anomalies across five categories with per-frame binary labels, plus real-world videos for sim-to-real transfer.","area":"Vision & 3D","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Automotive"],"capabilities":[],"topics":["cs.CV"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.31875","pdf":"https://arxiv.org/pdf/2606.31875","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.31875"},"evidence":{"snippet":"We introduce SENSE-VAD, the first synthetic video anomaly detection benchmark for autonomous driving explicitly designed around socially complex anomalies.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31875"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SENSE-VAD is a synthetic video anomaly detection dataset for autonomous driving, generated with CARLA and Unreal Engine. It includes socially complex anomalies across five categories with per-frame binary labels, plus real-world videos for sim-to-real transfer.","whyItMatters":"Addresses the evaluation gap for socially complex anomalies in autonomous driving, which are not captured by motion-based detectors. Provides a controlled benchmark to test current video anomaly detection models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"558112397f630815a69a6b85c4bfbbdd057e1e2c89c0dcf69cd808cd9ee19940"},"motivation":"Autonomous vehicles (AVs) must navigate not only motion-based hazards but also socially complex situations whose danger is constituted by inter-agent relationships rather than movement statistics alone.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31875","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_sentinelbench_48835b47","familyId":"bmf_f1b3e2bc0378","name":"SentinelBench","oneLine":"SentinelBench is an open-source benchmark for time-evolving monitoring tasks. It contains 100 tasks across 10 synthetic web environments (email, calendars, finance, etc.) with scripted events, measuring task completion, reaction time, and resource use.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05342","pdf":"https://arxiv.org/pdf/2606.05342","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05342"},"evidence":{"snippet":"To measure progress on this class of tasks, we introduce SentinelBench, an open-source benchmark for time-evolving monitoring tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05342"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SentinelBench is an open-source benchmark for time-evolving monitoring tasks. It contains 100 tasks across 10 synthetic web environments (email, calendars, finance, etc.) with scripted events, measuring task completion, reaction time, and resource use.","whyItMatters":"Long-running monitoring tasks require sustained attention rather than continuous action. SentinelBench captures this class and quantifies the tradeoff between responsiveness and cost, enabling comparison of agent designs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"91803cea33f9bfc490461b3f0abd19837a509a869515fef28f3e378465e4a650"},"motivation":"AI agents are increasingly asked to carry out work that spans minutes, hours, or longer.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05342","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_seq2synth_0ff26fb5","familyId":"bmf_a0760e0a5ca8","name":"Seq2Synth","oneLine":"Seq2Synth evaluates temporal fidelity of synthetic sequential tabular data across timestamp, cross-sectional, longitudinal, structural, and privacy dimensions, using a taxonomy to determine applicable metrics. It spans seven core datasets and multiple generators.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-07-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.15606","pdf":"https://arxiv.org/pdf/2607.15606","project":null,"code":"https://github.com/KiwanKwon/Seq2Synth","data":null,"hfPaper":"https://huggingface.co/papers/2607.15606"},"evidence":{"snippet":"We introduce Seq2Synth, a taxonomy-guided benchmark for evaluating whether synthetic sequential tabular data preserve these temporal structures.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.15606"},"ranking":{"90d":{"score":28,"rank":273,"coverage":0.55,"confidence":"Low"}},"description":"Seq2Synth evaluates temporal fidelity of synthetic sequential tabular data across timestamp, cross-sectional, longitudinal, structural, and privacy dimensions, using a taxonomy to determine applicable metrics. It spans seven core datasets and multiple generators.","whyItMatters":"Addresses the gap in evaluating temporal structure in synthetic tabular data, which static metrics miss, providing a standardized framework for researchers and practitioners.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2f587e57b042dc02da5ef87bc2b3c9955aec5c2e11ab2aed80564ecb083fa170"},"motivation":"Synthetic sequential tabular data are increasingly used for privacy-preserving data sharing and data-driven research, but evaluating their fidelity remains difficult because temporal structure is easily lost under conventional tabular metrics.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.15606","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_setoka_2726d0e3","familyId":"bmf_eec5c354913b","name":"Setoka","oneLine":"A benchmark for evaluating memory-augmented personalized agents on hierarchical user understanding from heterogeneous data, with four levels of user understanding and psychometrics-based synthetic data.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27056","pdf":"https://arxiv.org/pdf/2607.27056","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.27056"},"evidence":{"snippet":"In this work, we propose Setoka, a benchmark for evaluating memory-augmented personalized agents with hierarchical user understanding from heterogeneous data.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27056"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark for evaluating memory-augmented personalized agents on hierarchical user understanding from heterogeneous data, with four levels of user understanding and psychometrics-based synthetic data.","whyItMatters":"Personalized agents require deeper user understanding beyond fact retrieval; this benchmark provides a standard evaluation for cross-source integration and abstraction over long-term user behavior.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"709be3da37b7b0e950912e3ba256c14f0f48267d794cf70629dec6aa18a2ac20"},"motivation":"Personalized agents are increasingly applied to assist users across a wide range of tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27056","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_sevra-bench_b52ddf44","familyId":"bmf_74c5a4c5c66f","name":"SEVRA-BENCH","oneLine":"SEVRA-BENCH is a benchmark for measuring how often LLM-based code review agents approve adversarial pull requests with social-engineering framings, built from vulnerability-fixing commits. It includes a challenge split of roughly 1,000 adversarial PRs.","area":"Language & Knowledge","applicationDomains":["Cybersecurity","Robotics & Autonomous Systems"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity","Robotics"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.13757","pdf":"https://arxiv.org/pdf/2606.13757","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.13757"},"evidence":{"snippet":"We introduce SEVRA-BENCH (Social Engineering of Vulnerabilities in Review Agents), a benchmark that measures how often a review agent approves such adversarial PR s.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13757"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SEVRA-BENCH is a benchmark for measuring how often LLM-based code review agents approve adversarial pull requests with social-engineering framings, built from vulnerability-fixing commits. It includes a challenge split of roughly 1,000 adversarial PRs.","whyItMatters":"Review agents are susceptible to narrative manipulation, which can lead to merging vulnerable code. SEVRA-BENCH quantifies this gap in security capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f38cfe0a8590ef58503520196d8568252bb886e6eb913e6af5d30bbfbc991af0"},"motivation":"Large language models (LLMs) are increasingly deployed in automated code-review systems, where their approvals can determine which code is merged into shared repositories.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13757","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"bm_sgr-bench_d4ee6524","familyId":"bmf_3984c874ff5c","name":"SGR-Bench","oneLine":"SGR-Bench evaluates search agents on state-gated retrieval tasks. It includes 100 expert-curated tasks across 12 public data ecosystems, requiring agents to configure site-specific filters, views, hierarchies, or scopes to retrieve structured answers. Tasks come in goal-oriented and constraint-guided formulations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22219","pdf":"https://arxiv.org/pdf/2605.22219","project":null,"code":null,"data":"https://huggingface.co/datasets/PKUAIWeb/SGR-BENCH","hfPaper":"https://huggingface.co/papers/2605.22219"},"evidence":{"snippet":"We introduce SGR-Bench, a benchmark for this setting containing 100 expert-curated tasks spanning six source families and 12 public data ecosystems.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":190,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2605.22219"},"ranking":{},"description":"SGR-Bench evaluates search agents on state-gated retrieval tasks. It includes 100 expert-curated tasks across 12 public data ecosystems, requiring agents to configure site-specific filters, views, hierarchies, or scopes to retrieve structured answers. Tasks come in goal-oriented and constraint-guided formulations.","whyItMatters":"SGR-Bench addresses an undercharacterized class of retrieval tasks where evidence is hidden behind site-specific retrieval states. It provides a standardized evaluation to measure agents' ability to establish correct retrieval states, which is critical for real-world data retrieval from specialized websites.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"918576c8c087f4bc05fafe7baaf735ead5edcb0ab74b8f5ecfb8391d8c75d9a4"},"motivation":"Recent advances in large language models and tool-using agents have expanded the range of benchmarked web tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.22219","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"PKUAIWeb","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/datasets/PKUAIWeb/SGR-BENCH","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_shallowbench_e94b542b","familyId":"bmf_a1a0c0f21b0e","name":"ShallowBench","oneLine":"ShallowBench is a curated benchmark of 5,780 shallow-pocket targets for evaluating generative drug design models. Targets are extracted from CrossDocked2020 based on low concavity and sufficient surface area.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06717","pdf":"https://arxiv.org/pdf/2606.06717","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.06717"},"evidence":{"snippet":"To address this gap, we introduce ShallowBench, a strictly curated benchmark of 5,780 shallow-pocket targets extracted from CrossDocked2020.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06717"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ShallowBench is a curated benchmark of 5,780 shallow-pocket targets for evaluating generative drug design models. Targets are extracted from CrossDocked2020 based on low concavity and sufficient surface area.","whyItMatters":"Generative models often rely on deep pockets and struggle with shallow pockets. ShallowBench provides a testbed for developing models that can handle challenging targets.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"864ac842e6aeceb8e00eb2e46983f88c1f653e90e964c795b33a55d32d532b96"},"motivation":"While generative AI models have demonstrated remarkable success in structure-based drug design, they predominantly rely on deep binding pockets and struggle to sample effective ligands for challenging low-pocketability targets, such as the historically \"undruggable\" oncology targets KRAS and MYC.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.06717","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_shift-drift_175f8c38","familyId":"bmf_1080f520bb7a","name":"Shift & Drift","oneLine":"Shift & Drift evaluates closed-loop motion planners on two tracks: Semantic Shift, which uses a conversion pipeline to transform the DeepScenario Open 3D dataset into nuPlan for zero-shot testing on 1,182 scenarios across German cities and San Francisco, and State-Distribution Drift, which injects stochastic perturbations into ego-vehicle dynamics. Scoring is based on safety and progress metrics.","area":"Vision & 3D","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Automotive"],"capabilities":["Planning","Geometric reasoning"],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.07844","pdf":"https://arxiv.org/pdf/2607.07844","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.07844"},"evidence":{"snippet":"To address this, we present Shift & Drift, a novel dual-track benchmark designed to rigorously stress-test motion planners across two critical axes of distribution shift: (1) The Semantic Shift Track leverages a novel conversion pipeline that transforms the aerial, DeepScenario Open 3D dataset into the nuPlan simulation framework.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.07844"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Shift & Drift evaluates closed-loop motion planners on two tracks: Semantic Shift, which uses a conversion pipeline to transform the DeepScenario Open 3D dataset into nuPlan for zero-shot testing on 1,182 scenarios across German cities and San Francisco, and State-Distribution Drift, which injects stochastic perturbations into ego-vehicle dynamics. Scoring is based on safety and progress metrics.","whyItMatters":"Addresses the evaluation gap in generalization of motion planners to novel urban topologies and robustness to execution perturbations, providing a dual-track benchmark that quantifies the trade-off between imitation fidelity and closed-loop resilience.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"874d16e832f1ac20704d988aa859b1d5a60fb347a51d0047854909f0e8199d99"},"motivation":"While closed-loop motion planners trained on large-scale, object-level datasets, e.g., nuPlan, demonstrate strong in-distribution (ID) performance, their generalization to novel urban topologies and recovery mechanisms following execution perturbations remain under-explored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)","evidence":"Accepted at 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)","evidenceUrl":"https://arxiv.org/abs/2607.07844","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)","reviewStatus":"accepted","decisionRaw":"Accepted at 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.07844","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at 2026 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS 2026)","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_shopping-reasoning-bench_1ec16882","familyId":"bmf_241bcbe846af","name":"Shopping Reasoning Bench","oneLine":"Assesses multi-turn conversational shopping assistants across 525 expert-authored missions with importance-weighted binary rubrics. Evaluates reasoning across five categories covering preference refinement, trade-off analysis, and compatibility assessment.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-10","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.12608","pdf":"https://arxiv.org/pdf/2606.12608","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.12608"},"evidence":{"snippet":"We introduce the Shopping Reasoning Bench, an expert-authored benchmark of 525 missions (232 single-turn, 293 multi-turn) with 10863 importance-weighted binary rubrics authored by retail domain experts.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.12608"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Assesses multi-turn conversational shopping assistants across 525 expert-authored missions with importance-weighted binary rubrics. Evaluates reasoning across five categories covering preference refinement, trade-off analysis, and compatibility assessment.","whyItMatters":"Fills the lack of benchmarks for open-ended shopping dialogue with objective criteria and nuanced reasoning, providing a standardized testbed for improving assistant performance in real-world e-commerce.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"55844d5eb57bd09832f19037d0227541a00378d75d1ac069ce01f7cc260e7955"},"motivation":"Conversational shopping assistants now serve hundreds of millions of customers, yet no existing benchmark jointly evaluates the open-ended multi-turn reasoning, domain expertise, and criterion-level quality that real shopping conversations demand.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.12608","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Shopping Reasoning Bench Team","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/papers/2606.12608","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_shovir_1cca6b61","familyId":"bmf_72ae455766ea","name":"SHOVIR","oneLine":"SHOVIR evaluates vision shortcut learning in radiology report generation by extending MIMIC-CXR and PadChest-GR with per-box CheXpert labels. It defines image-level and disease-level occlusion experiments that compare model predictions on clean images against localized perturbations to isolate direct and contextual shortcut failures at the disease-class level.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.30201","pdf":"https://arxiv.org/pdf/2606.30201","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.30201"},"evidence":{"snippet":"We introduce SHOVIR, a benchmark for evaluating vision shortcut behavior in RRG.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.30201"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SHOVIR evaluates vision shortcut learning in radiology report generation by extending MIMIC-CXR and PadChest-GR with per-box CheXpert labels. It defines image-level and disease-level occlusion experiments that compare model predictions on clean images against localized perturbations to isolate direct and contextual shortcut failures at the disease-class level.","whyItMatters":"Standard RRG metrics miss whether diagnostic statements are grounded in actual image evidence, allowing models to exploit dataset priors. SHOVIR provides a protocol to assess spatial grounding, revealing that high report quality can coexist with shallow visual reliance, which is critical for clinical deployment decisions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"222d6f65b9d787c3b317be7a0386db381e68803608001c41ba22a0832d825ce7"},"motivation":"Current evaluation protocols for Vision-Language Models (VLMs) in Radiology Report Generation (RRG) rely on report-level metrics that measure lexical overlap or aggregate clinical correctness.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.30201","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_sidewalkbench_6d63d9b3","familyId":"bmf_8c660e63162e","name":"SidewalkBench","oneLine":"SidewalkBench evaluates visual navigation models on urban sidewalks using GPU-accelerated simulation with procedurally generated and real-world scanned scenes, including pedestrian-reactive and long-horizon scenarios.","area":"Robotics & Embodied AI","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.RO"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.16953","pdf":"https://arxiv.org/pdf/2606.16953","project":"https://vail-ucla.github.io/SidewalkBench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.16953"},"evidence":{"snippet":"To bridge this gap, we propose SidewalkBench, a comprehensive benchmark designed for visual navigation on urban sidewalks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16953"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SidewalkBench evaluates visual navigation models on urban sidewalks using GPU-accelerated simulation with procedurally generated and real-world scanned scenes, including pedestrian-reactive and long-horizon scenarios.","whyItMatters":"It addresses the lack of a unified benchmark for sidewalk navigation, providing a standardized simulation environment to compare model performance under realistic conditions, aiding progress in visual navigation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2e3985b22c49b0e6bbf9f5d0904396ad8408508ab9643fdbac56def25361b6b5"},"motivation":"Urban sidewalk navigation presents significant challenges due to complex structural layouts, dynamic pedestrian behaviors, and long distances.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.16953","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Vail UCLA","organizationType":"academic-lab","sourceUrl":"https://vail-ucla.github.io/SidewalkBench/","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"general"},{"id":"catalog_2c00e2642310ebbf","familyId":"catalog_family_2c00e2642310ebbf","name":"SIFO","oneLine":"SIFO (Simple Instruction Following) evaluates how well language models follow simple, explicit instructions. It tests fundamental instruction-following capabilities across various task types.","description":"SIFO (Simple Instruction Following) evaluates how well language models follow simple, explicit instructions. It tests fundamental instruction-following capabilities across various task types.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Instruction Following","Structured Output","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/sifo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2c00e2642310ebbf"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/sifo"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"sifo","url":"https://llm-stats.com/benchmarks/sifo","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["instruction following","structured output","general","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_8e550de1cf971aca","familyId":"catalog_family_8e550de1cf971aca","name":"SIFO-Multiturn","oneLine":"SIFO-Multiturn evaluates instruction following capabilities in multi-turn conversational settings, testing how well models maintain context and follow instructions across multiple exchanges.","description":"SIFO-Multiturn evaluates instruction following capabilities in multi-turn conversational settings, testing how well models maintain context and follow instructions across multiple exchanges.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Structured Output","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/sifo-multiturn","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8e550de1cf971aca"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/sifo-multiturn"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"sifo-multiturn","url":"https://llm-stats.com/benchmarks/sifo-multiturn","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["structured output","general","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_signpost-bench_c7de79e5","familyId":"bmf_dec2c32e4dca","name":"SIGNPOST-Bench","oneLine":"SIGNPOST-Bench evaluates text-vision conflict resolution in multimodal large language models via a counterfactual benchmark of image variants (Original, Blank, Similar, Random, Adversarial) for visual geolocation. It includes 5,111 groups and 25,555 variants, with metrics for localization error and target-directed shifts.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04244","pdf":"https://arxiv.org/pdf/2608.04244","project":null,"code":"https://github.com/inorganicwriter/SIGNPOST-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.04244"},"evidence":{"snippet":"We introduce SIGNPOST-Bench, a controlled counterfactual benchmark for evaluating text-vision conflict resolution.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":"2026-08-06T00:00:00.000Z","githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04244"},"ranking":{"30d":{"score":24,"rank":104,"coverage":0.85,"confidence":"High"},"90d":{"score":27,"rank":288,"coverage":0.7,"confidence":"Medium"}},"description":"SIGNPOST-Bench evaluates text-vision conflict resolution in multimodal large language models via a counterfactual benchmark of image variants (Original, Blank, Similar, Random, Adversarial) for visual geolocation. It includes 5,111 groups and 25,555 variants, with metrics for localization error and target-directed shifts.","whyItMatters":"Existing benchmarks rarely reveal how MLLMs arbitrate conflicting text and visual evidence. SIGNPOST-Bench provides a controlled framework to measure robustness to conflicting scene text, showing that localization performance degrades substantially under adversarial text edits.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8d133c017d3c6059a903e463c0466459c1859cd3a889e9bfb9adfce05d290ab5"},"motivation":"Multimodal large language models (MLLMs) make grounded predictions in real-world scenes by combining visual and textual cues, yet existing benchmarks rarely reveal how they arbitrate between these evidence sources when they conflict.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04244","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"inorganicwriter","organizationType":"community","sourceUrl":"https://github.com/inorganicwriter/SIGNPOST-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_silent-misresolution-benchmark_79846646","familyId":"bmf_39f93253928c","name":"Silent Misresolution","oneLine":"A benchmark for silent misresolution in dictation cleanup, where a system deletes spoken content without leaving a trace in the output; it includes a corpus with revision sites and a scorer for detecting such failures.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Software & Cloud","Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/hemachandra666/silent-misresolution-benchmark","pdf":null,"project":null,"code":"https://github.com/hemachandra666/silent-misresolution-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"silent-misresolution-benchmark A benchmark for silent misresolution in dictation cleanup: when a system deletes what you said and the output still reads perfectly.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:hemachandra666/silent-misresolution-benchmark"},"ranking":{"30d":{"score":23,"rank":140,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":344,"coverage":0.55,"confidence":"Low"}},"description":"A benchmark for silent misresolution in dictation cleanup, where a system deletes spoken content without leaving a trace in the output; it includes a corpus with revision sites and a scorer for detecting such failures.","whyItMatters":"Existing metrics like edit rate and WER are blind to silent misresolution because correct-looking outputs score well even when information is lost; this benchmark provides a targeted evaluation for a failure mode that is invisible to current quality processes.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"ae4e92e6a40c2113b13a092257571182445fd88d93d7ddf85f0909fe56455986"},"motivation":"silent-misresolution-benchmark A benchmark for silent misresolution in dictation cleanup: when a system deletes what you said and the output still reads perfectly.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The repository is newly created and serves primarily as a single paper's viewpoint probe rather than an established ongoing benchmark; there is no independent paper or official benchmark site referenced, and the corpus is small and narrowly scoped."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/hemachandra666/silent-misresolution-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"attentionForecast":{"score":0,"confidence":"Low","horizon":"7d","reason":"The benchmark addresses a niche dictation issue and lacks independent validation or a formal release context, limiting its likely early attention."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"specific"},{"id":"bm_silentbug-bench_29b06a59","familyId":"bmf_9af0df30b045","name":"silentbug","oneLine":"Evaluates agents on detecting silent ML training defects across 18 CPU-only PyTorch tasks, scoring patch application, metric recovery, localization, minimality, and false positives.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/Json604/silentbug-bench","pdf":null,"project":null,"code":"https://github.com/Json604/silentbug-bench","data":null,"hfPaper":null},"evidence":{"snippet":"silentbug-bench ML training defects that never crash.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:json604/silentbug-bench"},"ranking":{"30d":{"score":23,"rank":139,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":343,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates agents on detecting silent ML training defects across 18 CPU-only PyTorch tasks, scoring patch application, metric recovery, localization, minimality, and false positives.","whyItMatters":"Targets a neglected class of defects that do not crash but corrupt training, providing a detailed protocol for reproducible debugging evaluation.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"25c69576ef23b0e8839b78ea572984cfee4ff1107dd27ed3504f7178a9a00493"},"motivation":"silentbug-bench ML training defects that never crash.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/Json604/silentbug-bench","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"attentionForecast":{"score":50,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses an underexplored problem in ML debugging with clear evaluation mechanics, likely to interest researchers in code agents and ML engineering."},"evaluationMode":"score_submission","publishers":[{"name":"Json604","organizationType":"community","sourceUrl":"https://github.com/Json604/silentbug-bench","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_simmer_2d9d2c96","familyId":"bmf_365fcc9a87e9","name":"SIMMER","oneLine":"SIMMER evaluates latent failures in LLM-generated plans for kitchen-domain tasks using a curated symbolic world model with 77 actions and 262 objects, scoring error-free plans and latent hazard detection.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.14574","pdf":"https://arxiv.org/pdf/2606.14574","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.14574"},"evidence":{"snippet":"To address this gap, we introduce SIMMER, a benchmark for evaluating latent failures in LLM planning through a human-curated symbolic world model grounded in the kitchen domain.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.14574"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SIMMER evaluates latent failures in LLM-generated plans for kitchen-domain tasks using a curated symbolic world model with 77 actions and 262 objects, scoring error-free plans and latent hazard detection.","whyItMatters":"Existing plan benchmarks miss failures that don't immediately halt execution but compromise goals; SIMMER provides metrics for irreversible latent failures, important for safe deployment of LLM planners.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"354641bae5f4dfcc81b7856be20ae40c7a1448dfa8fe7042ce46c55500d3bc10"},"motivation":"Large language models (LLMs) are increasingly deployed as planners for autonomous agents in household environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"COLM 2026","evidence":"Accepted at COLM 2026","evidenceUrl":"https://arxiv.org/abs/2606.14574","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"COLM 2026","reviewStatus":"accepted","decisionRaw":"Accepted at COLM 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.14574","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at COLM 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_simple_0e42de19","familyId":"bmf_a7a39b72f297","name":"SIMPLE","oneLine":"SIMPLE is a simulation testbed for humanoid loco-manipulation with 60 tasks, 50 scenes, and over 1,000 objects, integrating data generation pipelines and benchmarking policies.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08278","pdf":"https://arxiv.org/pdf/2606.08278","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.08278"},"evidence":{"snippet":"To this end, we present SIMPLE, a unified simulation testbed for humanoid policy learning and evaluation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08278"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SIMPLE is a simulation testbed for humanoid loco-manipulation with 60 tasks, 50 scenes, and over 1,000 objects, integrating data generation pipelines and benchmarking policies.","whyItMatters":"Real-world humanoid evaluation is expensive; SIMPLE aims to provide a reproducible simulation benchmark, but public access to environments and data is not yet confirmed.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d25bde345149b06e4b4424f0ad30fcc3229dc65a3afa65bdbf407477681885a9"},"motivation":"Humanoid foundation models are advancing faster than we can evaluate them.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08278","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"lib_simpleqa","familyId":"family_simpleqa","name":"SimpleQA","oneLine":"Established benchmark family · Factuality & Grounding.","area":"Factuality & Grounding","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Factuality & Grounding"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://cdn.openai.com/papers/simpleqa.pdf","pdf":null,"project":"https://openai.com/index/introducing-simpleqa/","code":"https://github.com/openai/simple-evals","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_simpleqa"},"ranking":{},"recordType":"family","aliases":[],"sourceAttribution":[{"role":"official-release","url":"https://openai.com/index/introducing-simpleqa/"}],"adoptionRefs":["openai-gpt5"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"simpleQa","url":"https://benchlm.ai/benchmarks/simpleqa","paperUrl":"https://arxiv.org/abs/2411.04368","year":"2024","fullName":"Measuring Short-Form Factuality in Large Language Models","format":"Short-form Q&A","tasks":"Factual questions","successorKey":null},{"catalog":"llm-stats","sourceId":"simpleqa","url":"https://llm-stats.com/benchmarks/simpleqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","factuality","general"],"catalogModelCount":47,"catalogStarCount":0},{"id":"catalog_91f8e2135050765d","familyId":"catalog_family_91f8e2135050765d","name":"SimpleQA Verified","oneLine":"SimpleQA Verified is a curated, reliability-focused subset of SimpleQA that addresses label noise and redundancy in the original benchmark, measuring short-form parametric factual accuracy of large language models on fact-seeking questions with single, indisputable answers.","description":"SimpleQA Verified is a curated, reliability-focused subset of SimpleQA that addresses label noise and redundancy in the original benchmark, measuring short-form parametric factual accuracy of large language models on fact-seeking questions with single, indisputable answers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Factuality","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/simpleqa-verified","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_91f8e2135050765d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/simpleqa-verified"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"simpleqa-verified","url":"https://llm-stats.com/benchmarks/simpleqa-verified","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","factuality","general"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_3021d1189b97a933","familyId":"catalog_family_3021d1189b97a933","name":"SimpleVQA","oneLine":"SimpleVQA is a visual question answering benchmark focused on simple queries.","description":"SimpleVQA is a visual question answering benchmark focused on simple queries.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Image To Text","Multimodal","General","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3021d1189b97a933"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/simplevqa"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/simplevqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"simpleVqa","url":"https://benchlm.ai/benchmarks/simplevqa","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"SimpleVQA","format":"Image-grounded question answering","tasks":"Visual QA tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"simplevqa","url":"https://llm-stats.com/benchmarks/simplevqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","image to text","multimodal","general","vision"],"catalogModelCount":14,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_ef6a0d37e1015ec2","familyId":"catalog_family_ef6a0d37e1015ec2","name":"SingleCellBench","oneLine":"Single-cell RNA sequencing analysis tasks spanning common bioinformatics workflows.","description":"Single-cell RNA sequencing analysis tasks spanning common bioinformatics workflows.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ef6a0d37e1015ec2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/singlecellbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"singleCellBench","url":"https://benchlm.ai/benchmarks/singlecellbench","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"LatchBio SingleCellBench","format":"Task score","tasks":"195 single-cell RNA sequencing problems","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_fc60a2d907e5212e","familyId":"catalog_family_fc60a2d907e5212e","name":"Siren AgentDojo Attack Success Rate","oneLine":"Siren AgentDojo evaluates tool-using agents under adversarial prompt-injection attacks. This metric is the attack success rate; lower is better.","description":"Siren AgentDojo evaluates tool-using agents under adversarial prompt-injection attacks. This metric is the attack success rate; lower is better.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Safety","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/siren-agentdojo-attack-success","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fc60a2d907e5212e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/siren-agentdojo-attack-success"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"siren-agentdojo-attack-success","url":"https://llm-stats.com/benchmarks/siren-agentdojo-attack-success","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety","agents","tool calling"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_5fb391ffe84c3f24","familyId":"catalog_family_5fb391ffe84c3f24","name":"Siren AgentDojo Utility","oneLine":"Siren AgentDojo evaluates tool-using agents under adversarial prompt-injection attacks. This metric reports utility on the assigned tasks.","description":"Siren AgentDojo evaluates tool-using agents under adversarial prompt-injection attacks. This metric reports utility on the assigned tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Safety","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/siren-agentdojo-utility","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5fb391ffe84c3f24"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/siren-agentdojo-utility"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"siren-agentdojo-utility","url":"https://llm-stats.com/benchmarks/siren-agentdojo-utility","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety","agents","tool calling"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_siren-bench_1d0cc158","familyId":"bmf_94eafd231d5e","name":"SIREN-Bench-v1","oneLine":"SIREN-Bench-v1 is a benchmark for emergency-vehicle (EMV) interactions, built on a SUMO-CARLA co-simulation platform. It includes seven parameterized interaction templates across emergency levels L1-L3 and three behavior families, with synchronized sensor observations and simulator-native annotations. Supports tasks such as 3D object detection, trajectory prediction, and vision-language risk understanding.","area":"Robotics & Embodied AI","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.RO"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-26","recognitionConfidence":0.65,"links":{"report":"https://arxiv.org/abs/2608.24094","pdf":"https://arxiv.org/pdf/2608.24094","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We instantiate the platform as \\textbf{SIREN-Bench-v1}, comprising seven parameterized interaction templates across emergency levels L1--L3 and three behavior families, with synchronized sensor observations and simulator-native annotations.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24094"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SIREN-Bench-v1 is a benchmark for emergency-vehicle (EMV) interactions, built on a SUMO-CARLA co-simulation platform. It includes seven parameterized interaction templates across emergency levels L1-L3 and three behavior families, with synchronized sensor observations and simulator-native annotations. Supports tasks such as 3D object detection, trajectory prediction, and vision-language risk understanding.","whyItMatters":"Evaluating EMV interactions requires behavior-level control over both EMV privileges and civilian responses, which existing benchmarks lack. SIREN-Bench-v1 provides a reproducible co-simulation environment with behavior-dependent failure modes, enabling standard evaluation for safety-critical autonomous driving research.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"f5d6159b18985b0bac6cd821f18230ca15e8b68bd4bd2669ebec8b5d755d2f69"},"motivation":"Emergency vehicles (EMVs) can reorganize surrounding traffic as civilian vehicles brake, change lanes, or form rescue corridors in response to their passage.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:08:47.245811Z","model":"deepseek-v4-pro","decisionReason":"Benchmark defines a repeatable evaluation platform with parameterized templates and evaluated tasks, and public release is stated in the abstract.","canonicalNameSource":"paper_title","canonicalNameEvidence":"SIREN-Bench: Behavior-Driven Generation and Evaluation of Emergency-Vehicle Interactions"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24094","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":75,"confidence":"Medium","horizon":"7d","reason":"Addresses safety-critical autonomous driving with a unique co-simulation approach, likely to attract attention from robotics and transportation researchers."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"general"},{"id":"bm_sis-bench_e31a9de3","familyId":"bmf_067ed93fb5e8","name":"SIS-Bench","oneLine":"SIS-Bench evaluates embodied spatial intelligence in UAV scenarios across two dimensions (spatial cognition and self-awareness) and three cognitive levels (perception, memory, reasoning), with 4,856 QA pairs from 1,646 real-world UAV videos.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.CV"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.12477","pdf":"https://arxiv.org/pdf/2607.12477","project":"https://choucisan.github.io/publications/self-in-space","code":"https://github.com/IntelliSensing/Self-in-Space","data":null,"hfPaper":"https://huggingface.co/papers/2607.12477"},"evidence":{"snippet":"To address this gap, we introduce SIS-Bench, a benchmark for evaluating embodied spatial intelligence in UAV scenarios under a unified self-in-space formulation.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":11,"hfDailySubmittedAt":null,"githubStars":33,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.12477"},"ranking":{"90d":{"score":48,"rank":83,"coverage":0.7,"confidence":"Medium"}},"description":"SIS-Bench evaluates embodied spatial intelligence in UAV scenarios across two dimensions (spatial cognition and self-awareness) and three cognitive levels (perception, memory, reasoning), with 4,856 QA pairs from 1,646 real-world UAV videos.","whyItMatters":"Existing UAV benchmarks are environment-centric, leaving agent self-awareness implicit. SIS-Bench provides a unified self-in-space evaluation protocol, revealing imbalances between spatial and self-related cognition and offering practical value for developing UAV embodied models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f1d33b45c10ac866cad55c10908802ca03020edc64a37af3818fb8104b29aca9"},"motivation":"Autonomous UAV systems increasingly rely on multimodal large language models (MLLMs) to operate in complex real-world environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.12477","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"IntelliSensing","organizationType":"academic-lab","sourceUrl":"https://github.com/IntelliSensing/Self-in-Space","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_skill-use_dfe395c1","familyId":"bmf_43888fe52bed","name":"Skill-Use","oneLine":"Skill-Use evaluates skill use in agentic harnesses through 79 real skills and 177 executable tasks across nine domains, measuring trigger, compliance, and boundary adherence with a combined SU score.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04828","pdf":"https://arxiv.org/pdf/2608.04828","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.04828"},"evidence":{"snippet":"We introduce Skill-Use, a benchmark that evaluates skill use under progressive disclosure, where an agent sees only a skill's name and short description and must retrieve the full procedure before following it.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04828"},"ranking":{"30d":{"score":39,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Skill-Use evaluates skill use in agentic harnesses through 79 real skills and 177 executable tasks across nine domains, measuring trigger, compliance, and boundary adherence with a combined SU score.","whyItMatters":"Agent evaluations often focus on task success, not whether agents can autonomously identify and apply relevant skills. Skill-Use isolates skill retrieval and usage under progressive disclosure, showing harness-dependent capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6a7c1ac3397cd3107e6e682fc0e118b50f3b5b22a71cde757279c5a932636cb0"},"motivation":"Large language model (LLM) agents increasingly rely on skills, structured documents that specify when to act, which procedure to follow, and which tools are allowed.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04828","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_skillchain-gym_68720c0a","familyId":"bmf_b51f64778801","name":"SkillChain-Gym","oneLine":"SkillChain-Gym evaluates production-inventory control policies with reskilling dynamics, including skill certification, forgetting, and training constraints. Metrics cover operations, resilience, capability growth, and training access across seeded disruption scenarios.","area":"Language & Knowledge","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Logistics"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.17266","pdf":"https://arxiv.org/pdf/2606.17266","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.17266"},"evidence":{"snippet":"We introduce SkillChain-Gym, a benchmark specification for reskilling-aware production-inventory control: a single-site environment with stylized worker skill-state dynamics, hard threshold certification, forgetting, and capacity-consuming training actions constrained by the same per-worker time budget as production.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.17266"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SkillChain-Gym evaluates production-inventory control policies with reskilling dynamics, including skill certification, forgetting, and training constraints. Metrics cover operations, resilience, capability growth, and training access across seeded disruption scenarios.","whyItMatters":"Workforce skills are often overlooked in production benchmarks. SkillChain-Gym provides a testbed to evaluate adaptive policies under skill dynamics, aiding in workforce planning and operational resilience.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"d9ad4f9e8e34cfcee34be6687be53b2988baec92e3e6338d8ebbd1a5984c72df"},"motivation":"Production planning increasingly has to treat workforce capability as a decision variable: certifications lapse when skills are not maintained, new products require skills the current workforce does not hold, and reskilling competes for the same worker hours needed for production.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.17266","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_skillevolbench_997ef790","familyId":"bmf_0d32e638a5b1","name":"SkillEvolBench","oneLine":"SkillEvolBench evaluates whether LLM agents can distill episodic experience into reusable procedural skills. It includes 180 tasks across six environments, with acquisition tasks and frozen deployment tasks testing context shift, adversarial shortcuts, and composition. Scoring uses success rates and other agent metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24117","pdf":"https://arxiv.org/pdf/2605.24117","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24117"},"evidence":{"snippet":"We introduce SkillEvolBench, a diagnostic benchmark for evaluating this step from experience reuse to skill formation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":22,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24117"},"ranking":{},"description":"SkillEvolBench evaluates whether LLM agents can distill episodic experience into reusable procedural skills. It includes 180 tasks across six environments, with acquisition tasks and frozen deployment tasks testing context shift, adversarial shortcuts, and composition. Scoring uses success rates and other agent metrics.","whyItMatters":"It is unclear whether LLM agents can form durable procedural skills from experience. SkillEvolBench provides a diagnostic testbed comparing skill-based learning against raw-trajectory reuse, informing agent design and training.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"67b22643bad860b49d7efe0353d87e1c99be64e10066d727165156af7747dd3c"},"motivation":"Large language model (LLM) agents accumulate rich episodic trajectories while solving real-world tasks, but it remains unclear whether such experience can be distilled into reusable procedural skills.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24117","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_skillharm_560fd4c5","familyId":"bmf_8422ad323405","name":"SkillHarm","oneLine":"Benchmark of skill-based attacks across the skill-use lifecycle, with 879 attack samples across 71 skills, evaluating Fixed-Payload Poisoning and Self-Mutating Poisoning scenarios across 12 risk types, with attack success rate as primary metric.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02540","pdf":"https://arxiv.org/pdf/2606.02540","project":null,"code":"https://github.com/OSU-NLP-Group/SkillHarm","data":null,"hfPaper":"https://huggingface.co/papers/2606.02540"},"evidence":{"snippet":"To bridge these gaps, we introduce SkillHarm, a benchmark of skill-based attacks across the skill-use lifecycle, paired with a systematic taxonomy of skill-relevant risks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":10,"hfDailySubmittedAt":null,"githubStars":10,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02540"},"ranking":{},"description":"Benchmark of skill-based attacks across the skill-use lifecycle, with 879 attack samples across 71 skills, evaluating Fixed-Payload Poisoning and Self-Mutating Poisoning scenarios across 12 risk types, with attack success rate as primary metric.","whyItMatters":"Existing studies evaluate poisoned skills within single task executions and use ad-hoc risk lists. This benchmark systematically covers lifecycle-aware attacks and provides a taxonomy and construction pipeline for reproducible evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d9bf68d429d90f900361bb66d90fcee3ebc365ed25421becd42c88b9fb1e5dc7"},"motivation":"Agent skills occupy a privileged position in the agent workflow, as agents are expected to implicitly follow and execute them, rendering third-party skills a vulnerable attack surface.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02540","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"OSU-NLP-Group","organizationType":"academic-lab","sourceUrl":"https://github.com/OSU-NLP-Group/SkillHarm","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_practice-makes-unsafe-skill-misevolution-i_3d2c225f","familyId":"bmf_be0110c7fe59","name":"SkillMisevo-Bench","oneLine":"Lifecycle-aware benchmark for persistent safety failures in self-improving LLM agents, consisting of 25 episodes with malicious demonstrations, benign twins, and fresh-session probes, scored with nine metrics including carryover ASR and unsafe retrieval.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.12851","pdf":"https://arxiv.org/pdf/2608.12851","project":null,"code":"https://github.com/henrymao2004/misevolve","data":null,"hfPaper":null},"evidence":{"snippet":"To expose this lifecycle, we introduce SkillMisevo-Gym, a lifecycle-aware harness that versions skill state across agent frameworks, and SkillMisevo-Bench, a frozen design from malicious exposure to carryover tasks, with concept-aligned benign tasks and nine lifecycle metrics.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12851"},"ranking":{"30d":{"score":38,"rank":53,"coverage":0.55,"confidence":"Low"},"90d":{"score":37,"rank":172,"coverage":0.55,"confidence":"Low"}},"description":"Lifecycle-aware benchmark for persistent safety failures in self-improving LLM agents, consisting of 25 episodes with malicious demonstrations, benign twins, and fresh-session probes, scored with nine metrics including carryover ASR and unsafe retrieval.","whyItMatters":"Exposes how unsafe experiences can become reusable policies that cause later harm, enabling measurement of risk across skill authoring, retrieval, and reuse.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"785db45ce0c88fa23dc05edc0e20cfda2a742d55ee5cf0d7b8844b696afecb2b"},"motivation":"Self-improving LLM agents convert successful trajectories into persistent cross-task state.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The frozen benchmark design with clear episodes and metrics provides a stable scoring contract, and the repository includes code and evaluation procedures for public reuse.","canonicalNameSource":"abstract","canonicalNameEvidence":"SkillMisevo-Bench, a frozen design from malicious exposure to carryover tasks, with concept-aligned benign tasks and nine lifecycle metrics"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12851","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":58,"confidence":"Medium","horizon":"7d","reason":"The safety-focused angle and provided code repository should drive early interest among agent safety researchers."},"evaluationMode":"public_reusable","publishers":[{"name":"Henry Mao and collaborators","organizationType":"academic-lab","sourceUrl":"https://github.com/henrymao2004/misevolve","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_skillreason-bench_12813801","familyId":"bmf_71b30f2ba1c9","name":"SkillReason-Bench","oneLine":"SkillReason-Bench is a retrieval benchmark with 3,729 queries and 61,228 skills across nine domains, but the paper focuses on proposing a retrieval method (SkillReason) rather than the benchmark itself.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Information retrieval"],"topics":["Agents","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.08640","pdf":"https://arxiv.org/pdf/2608.08640","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.08640"},"evidence":{"snippet":"To address this gap, we introduce SkillReason-Bench, a large-scale cross-domain benchmark containing 3,729 queries and a retrieval corpus of 61,228 skills spanning nine domains.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08640"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SkillReason-Bench is a retrieval benchmark with 3,729 queries and 61,228 skills across nine domains, but the paper focuses on proposing a retrieval method (SkillReason) rather than the benchmark itself.","whyItMatters":"The benchmark serves as a testbed for the proposed method and is compared to existing benchmarks, but the primary contribution is the method, not a standalone reusable benchmark.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"21f818c80bbcc303d0410a059f0296a48bab3d92a36efb6edb9bc0964b74ec47"},"motivation":"Large language model agents increasingly rely on reusable skills to extend their capabilities beyond parametric knowl- edge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.08640","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_skillresolve-bench_99845262","familyId":"bmf_df773b457612","name":"SkillResolve-Bench","oneLine":"SkillResolve-Bench 1.0 evaluates agent skill retrieval under same-capability ambiguity, pairing helpful skills with risky siblings. It includes 661 pairs, a 7,982-candidate pool, disjoint splits, and reports Recall@K and harmful sibling rate (HSR@K).","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.10388","pdf":"https://arxiv.org/pdf/2606.10388","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.10388"},"evidence":{"snippet":"We introduce SkillResolve-Bench 1.0, an auditable benchmark for this setting with 661 helpful/risky pairs, source-role and admission evidence, cue/leakage checks, query-disjoint splits, and a 7,982-candidate pool that includes 6,660 public SkillRet candidates.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.10388"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SkillResolve-Bench 1.0 evaluates agent skill retrieval under same-capability ambiguity, pairing helpful skills with risky siblings. It includes 661 pairs, a 7,982-candidate pool, disjoint splits, and reports Recall@K and harmful sibling rate (HSR@K).","whyItMatters":"Skill retrieval carries execution risk beyond relevance. This benchmark quantifies exposure to risky siblings, supporting development of retrievers that select safe representatives, reducing harmful failures in agent deployments.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a39b09a08c49f30abc0881fb8d26b02163a742f40d34c9d96279d74301e94972"},"motivation":"Agent skill libraries are becoming routable software assets: a retrieved skill can contribute instructions, scripts, resource bindings, and execution assumptions to an agent.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.10388","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_skillsafe-bench_e266705a","familyId":"bmf_1ffd6ecd5782","name":"SkillSafe-Bench","oneLine":"SkillSafe-Bench evaluates skill-merged LLMs on static refusal, adaptive jailbreak robustness, and capability retention using a two-judge AND rule, across multiple open-weight bases and attack types.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Robustness"],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.08542","pdf":"https://arxiv.org/pdf/2608.08542","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.08542"},"evidence":{"snippet":"We introduce SkillSafe-Bench, a controlled benchmark that scores skill-merged models on static refusal, adaptive jailbreak robustness, and capability retention under a conservative two-judge AND rule.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08542"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SkillSafe-Bench evaluates skill-merged LLMs on static refusal, adaptive jailbreak robustness, and capability retention using a two-judge AND rule, across multiple open-weight bases and attack types.","whyItMatters":"It exposes that static safety does not predict robustness to adaptive attacks, providing a more accurate safety evaluation for model merging and guiding safer merging practices.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cde271257299f16d1e2889b842f9edf7caa9647aa300443ef945774ed634650d"},"motivation":"Model merging has become the default way to give an aligned language model new skills without retraining: a practitioner folds task vectors from math, code, or domain specialists into a safety-aligned base using task arithmetic, TIES, or DARE.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.08542","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_f0e708e2d445b453","familyId":"catalog_family_f0e708e2d445b453","name":"SkillsBench","oneLine":"SkillsBench evaluates coding agents on self-contained programming tasks, measuring practical engineering skills across diverse software development scenarios.","description":"SkillsBench evaluates coding agents on self-contained programming tasks, measuring practical engineering skills across diverse software development scenarios.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/skillsbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f0e708e2d445b453"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/skillsbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/skillsbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"skillsBench","url":"https://benchlm.ai/benchmarks/skillsbench","paperUrl":"https://www.vals.ai/benchmarks/skillsbench","year":"2026","fullName":"Vals SkillsBench","format":"Accuracy score","tasks":"Agent skill-importance tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"skillsbench","url":"https://llm-stats.com/benchmarks/skillsbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["external","agents","code"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_skillvetbench_809c89ae","familyId":"bmf_13b32737c884","name":"SkillVetBench","oneLine":"SkillVetBench is a two-stage security vetting benchmark for open agentic skill ecosystems, evaluating detection of malicious skills via semantic analysis and runtime verification in a sandbox. The benchmark is built from confirmed malicious skills in the OpenClaw ecosystem.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00925","pdf":"https://arxiv.org/pdf/2606.00925","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.00925"},"evidence":{"snippet":"We present SkillVetBench, a two-stage security vetting benchmark for open agentic skill ecosystems.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00925"},"ranking":{},"description":"SkillVetBench is a two-stage security vetting benchmark for open agentic skill ecosystems, evaluating detection of malicious skills via semantic analysis and runtime verification in a sandbox. The benchmark is built from confirmed malicious skills in the OpenClaw ecosystem.","whyItMatters":"Open agent platforms face supply-chain risks from malicious skills, but existing defenses lack a standardized evaluation. This benchmark addresses the gap in measuring both detection and runtime verification, offering practical value for improving security in agent ecosystems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6cd31b8641e63a6bd86c0c4d989c9c886dbfffc6108291f43641b0a1800463bb"},"motivation":"Open agent platforms allow community contributors to publish reusable skills that agents can invoke at runtime.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00925","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_skysealand_02e2341b","familyId":"bmf_9cf9f27a7978","name":"SkySeaLand","oneLine":"SkySeaLand is a satellite object detection dataset with 1,307 high-resolution images and 19,101 bounding boxes across four classes. It provides COCO and YOLO annotations, a common split, and COCO metrics for detector evaluation.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.07382","pdf":"https://arxiv.org/pdf/2608.07382","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07382"},"evidence":{"snippet":"SkySeaLand provides a compact benchmark for mixed land--maritime transportation detection, while SkyDet establishes a documented low-footprint reference rather than a state-of-the-art accuracy claim.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07382"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SkySeaLand is a satellite object detection dataset with 1,307 high-resolution images and 19,101 bounding boxes across four classes. It provides COCO and YOLO annotations, a common split, and COCO metrics for detector evaluation.","whyItMatters":"The benchmark addresses a gap in wide-format satellite imagery detection, offering a compact dataset with a fixed protocol to compare detectors under standard metrics, supporting practical model selection for transportation monitoring.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7daac9c2572581f44e854a045803e215d0c3b2717b75e62d2df5212cb3522fd1"},"motivation":"Satellite object detection is challenged by small targets and wide-format scenes that lose detail under standard square-input resizing.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07382","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_4731160c7b1e1f07","familyId":"catalog_family_4731160c7b1e1f07","name":"SlakeVQA","oneLine":"A semantically-labeled knowledge-enhanced dataset for medical visual question answering. Contains 642 radiology images (CT scans, MRI scans, X-rays) covering five body parts and 14,028 bilingual English-Chinese question-answer pairs annotated by experienced physicians. Features comprehensive semantic labels and a structural medical knowledge base with both vision-only and knowledge-based questions requiring external medical knowledge reasoning.","description":"A semantically-labeled knowledge-enhanced dataset for medical visual question answering. Contains 642 radiology images (CT scans, MRI scans, X-rays) covering five body parts and 14,028 bilingual English-Chinese question-answer pairs annotated by experienced physicians. Features comprehensive semantic labels and a structural medical knowledge base with both vision-only and knowledge-based questions requiring external medical knowledge reasoning.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Image To Text","Multimodal","Reasoning","Healthcare","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/slakevqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4731160c7b1e1f07"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/slakevqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"slakevqa","url":"https://llm-stats.com/benchmarks/slakevqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image to text","multimodal","reasoning","healthcare","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_slapbench_99d61c1b","familyId":"bmf_d255d1cf42d2","name":"SLAPBench","oneLine":"SLAPBench evaluates multimodal LLMs on four-finger SLAP fingerprint verification using NIST SD302b, with 7,832 image pairs. It assesses identity verification under zero-shot, task-description, and similarity-scoring prompts, reporting metrics like AUC and FAR.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.15517","pdf":"https://arxiv.org/pdf/2607.15517","project":null,"code":"https://github.com/bibeshpyakurel/SLAPBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.15517"},"evidence":{"snippet":"We introduce SLAPBench, the first benchmark for MLLM-based four-finger SLAP fingerprint verification, built from NIST SD302b with 7,832 pairs (176 mated, 7,656 non-mated).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.15517"},"ranking":{"90d":{"score":23,"rank":370,"coverage":0.55,"confidence":"Low"}},"description":"SLAPBench evaluates multimodal LLMs on four-finger SLAP fingerprint verification using NIST SD302b, with 7,832 image pairs. It assesses identity verification under zero-shot, task-description, and similarity-scoring prompts, reporting metrics like AUC and FAR.","whyItMatters":"First benchmark for MLLM-based SLAP fingerprint verification, enabling systematic comparison of models and prompting strategies for biometric identity verification tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5525872300f07a8a02b953c4bd78003e244a158b531bc099345a0ba50035a390"},"motivation":"Four-finger SLAP fingerprints are flat live-scan impressions of the index, middle, ring, and little fingers of one hand, used for identity verification in border control and law enforcement.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.15517","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_slvmbench_4b061162","familyId":"bmf_6aaeb103fa63","name":"SLVMBench","oneLine":"SLVMBench evaluates video-LLMs' ability to learn skills from long video memory and apply them to real-time tasks, using 2-3 hour streams with embedded tutorials and human-annotated questions.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.11312","pdf":"https://arxiv.org/pdf/2607.11312","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.11312"},"evidence":{"snippet":"We introduce Skill Learning from Video Memory (SLVMBench), the first benchmark that jointly evaluates whether video large language models (video-LLMs) can learn skills from long video memory and apply them to real-time tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.11312"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SLVMBench evaluates video-LLMs' ability to learn skills from long video memory and apply them to real-time tasks, using 2-3 hour streams with embedded tutorials and human-annotated questions.","whyItMatters":"This is the first benchmark to test skill learning from long-context video memory, revealing significant limitations in current video LLMs and providing a realistic evaluation for skill acquisition.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"89196b25fda0c9dce27ac8b3c81878701042f5293681496027a0aeb1a311ba44"},"motivation":"We introduce Skill Learning from Video Memory (SLVMBench), the first benchmark that jointly evaluates whether video large language models (video-LLMs) can learn skills from long video memory and apply them to real-time tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.11312","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_smellbench_001e20cd","familyId":"bmf_d68b77fdba3e","name":"SmellBench","oneLine":"SmellBench is a code refactoring benchmark that proactively injects code smells into clean code snippets. It contains 294 cases across 7 smell types, 3 difficulty levels, and 2 instruction settings, with evaluation covering functional correctness, localization, and refactoring quality.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05574","pdf":"https://arxiv.org/pdf/2606.05574","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05574"},"evidence":{"snippet":"In this paper, we propose SmellBench, an extensible code refactoring benchmark that proactively injects code smells into clean code snippets from real-world repositories.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05574"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SmellBench is a code refactoring benchmark that proactively injects code smells into clean code snippets. It contains 294 cases across 7 smell types, 3 difficulty levels, and 2 instruction settings, with evaluation covering functional correctness, localization, and refactoring quality.","whyItMatters":"Existing benchmarks focus on functional correctness, not long-term maintainability. SmellBench could evaluate code agents' ability to produce maintainable code.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"94c03122ac4db821a90f20da7ea46bef2eb8fc4b0d1eee8d23af746250dd6e73"},"motivation":"Code Agents have achieved remarkable advances in recent years, exhibiting strong capabilities across a wide range of software engineering tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05574","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_smh-bench_ef4613ad","familyId":"bmf_eb0612cdd00c","name":"SMH-Bench","oneLine":"Benchmark for LLM agents in smart-home environments, with 1,100 tasks across 7 categories and 22 subcategories, built on the executable HomeEnv simulator, stratified by home complexity.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01912","pdf":"https://arxiv.org/pdf/2606.01912","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01912"},"evidence":{"snippet":"To address these limitations, we introduce SMH-Bench, a comprehensive benchmark for evaluating LLMs in smart-home environments.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01912"},"ranking":{},"description":"Benchmark for LLM agents in smart-home environments, with 1,100 tasks across 7 categories and 22 subcategories, built on the executable HomeEnv simulator, stratified by home complexity.","whyItMatters":"Addresses deficiencies in existing smart-home benchmarks by evaluating state-dependent reasoning across devices, but lacks a public comparison path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"29c5a612b2da0a8f544596988e46f44d3b446c5c05c52d96f065c4d053529a2c"},"motivation":"Smart homes are evolving toward complex state-dependent living environments, requiring Large Language Models (LLMs) to reason over user intent, preferences, and multi-device interactions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01912","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_a79cd338bc95571f","familyId":"catalog_family_a79cd338bc95571f","name":"SOB Value Acc","oneLine":"A structured-output benchmark from Interfaze measuring whether extracted JSON leaf values exactly match verified ground truth.","description":"A structured-output benchmark from Interfaze measuring whether extracted JSON leaf values exactly match verified ground truth.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Instructionfollowing"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://interfaze.ai/leaderboards/structured-output-benchmark","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a79cd338bc95571f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/sobvalueacc"}],"catalogSources":[{"catalog":"benchlm","sourceId":"sobValueAcc","url":"https://benchlm.ai/benchmarks/sobvalueacc","paperUrl":"https://interfaze.ai/leaderboards/structured-output-benchmark","year":"2026","fullName":"Structured Output Benchmark Value Accuracy","format":"Value accuracy","tasks":"Structured output extraction","successorKey":null}],"catalogCategories":["instructionFollowing"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_cdf285c8d0c2503a","familyId":"catalog_family_cdf285c8d0c2503a","name":"Social IQa","oneLine":"The first large-scale benchmark for commonsense reasoning about social situations. Contains 38,000 multiple choice questions probing emotional and social intelligence in everyday situations, testing commonsense understanding of social interactions and theory of mind reasoning about the implied emotions and behavior of others.","description":"The first large-scale benchmark for commonsense reasoning about social situations. Contains 38,000 multiple choice questions probing emotional and social intelligence in everyday situations, testing commonsense understanding of social interactions and theory of mind reasoning about the implied emotions and behavior of others.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Psychology","Reasoning","Creativity"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/social-iqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_cdf285c8d0c2503a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/social-iqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"social-iqa","url":"https://llm-stats.com/benchmarks/social-iqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["psychology","reasoning","creativity"],"catalogModelCount":9,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_socialpersona_4370a8a3","familyId":"bmf_6f5f69e2ce5e","name":"SocialPersona","oneLine":"SocialPersona evaluates multimodal LLMs on recovering revealed preferences from longitudinal social-media timelines and using them in dialogue. It includes 171 user timelines, 2,597 human-verified preference tags across seven domains, and supports profile construction and response generation tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26654","pdf":"https://arxiv.org/pdf/2606.26654","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.26654"},"evidence":{"snippet":"We introduce SocialPersona, a benchmark for evaluating whether multimodal large language models (MLLMs) can recover revealed preferences from longitudinal social-media timelines and use them in dialogue.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26654"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SocialPersona evaluates multimodal LLMs on recovering revealed preferences from longitudinal social-media timelines and using them in dialogue. It includes 171 user timelines, 2,597 human-verified preference tags across seven domains, and supports profile construction and response generation tasks.","whyItMatters":"Existing personalization benchmarks rely on explicitly stated preferences; SocialPersona tests inference from natural multimodal traces, a harder capability. It provides a reusable benchmark for measuring progress on long-horizon user modeling and personalized dialogue.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"47008b2a46569b0f3b0a4421f92d644c779fa1404b9c51fb27ca0fba3eb377bb"},"motivation":"Personalized language-model assistants are often evaluated through a memory lens: can a model recall preferences users have explicitly stated in dialogue?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.26654","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_societybench_8a44f535","familyId":"bmf_2d0da2d0dcde","name":"SocietyBench","oneLine":"SocietyBench is a benchmark for forecasting counterfactual social-world evolution, collecting web news and social-media posts to build timelines and generating forecasting questions scored on probability calibration and temporal accuracy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.04009","pdf":"https://arxiv.org/pdf/2608.04009","project":"https://co-minder.github.io/Societybench","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.04009"},"evidence":{"snippet":"We introduce SocietyBench, an end-to-end benchmark that takes a one-line event topic, collects Web news and social-media posts across five platforms, distills them into a date-indexed timeline that keeps factual events and a public-opinion layer separate, and then turns every cutoff date on that timeline into an audited bank of forecasting questions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.04009"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SocietyBench is a benchmark for forecasting counterfactual social-world evolution, collecting web news and social-media posts to build timelines and generating forecasting questions scored on probability calibration and temporal accuracy.","whyItMatters":"Current benchmarks focus on task completion, not on social understanding and forecasting. SocietyBench measures how well models understand and predict social events, providing a complementary evaluation axis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c56bd9b46af2a4422d3b738fef97316b2f035d9027b08c18675b5364c001d03e"},"motivation":"Large language models (LLMs), and the agents built on top of them, are now benchmarked heavily on whether they can finish a task -- fix a bug, drive a browser, operate a GUI.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.04009","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_soco_b681a0c6","familyId":"bmf_1ea2d8f8a54b","name":"SOCO","oneLine":"SOCO is a benchmark for semantic object correspondence with keypoint annotations across 100 categories and over 1M pairs. The evaluation is implemented within OmniProbe, which includes correspondence tasks among others.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.31597","pdf":"https://arxiv.org/pdf/2605.31597","project":"https://genintel.github.io/SOCO/","code":"https://github.com/GenIntel/OmniProbe","data":null,"hfPaper":"https://huggingface.co/papers/2605.31597"},"evidence":{"snippet":"To enable a systematic SC evaluation, we introduce SOCO, a new benchmark for Semantic Object Correspondence that introduces a taxonomy of correspondence types and provides consistent, functionally meaningful keypoint annotations across 100 categories and over 1M correspondence pairs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":null,"githubStars":11,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.31597"},"ranking":{},"description":"SOCO is a benchmark for semantic object correspondence with keypoint annotations across 100 categories and over 1M pairs. The evaluation is implemented within OmniProbe, which includes correspondence tasks among others.","whyItMatters":"It provides a systematic evaluation for part-level understanding in vision models, but since it is part of a larger framework, it is not a standalone benchmark.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c1a7d3d65c4c02c3613c3ae5b1a6a2546f53601b196660c339cd8dc3ce1fdf65"},"motivation":"Measuring structured object understanding in vision foundation models remains challenging due to inconsistent evaluation protocols and limited part-level supervision.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.31597","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"GenIntel","organizationType":"academic-lab","sourceUrl":"https://github.com/GenIntel/OmniProbe","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_socrates_dac2012d","familyId":"bmf_710482eaa280","name":"SoCRATES","oneLine":"SoCRATES is a benchmark for evaluating proactive LLM mediators in realistic, multi-domain testbeds. It contains 600 conflict scenarios across eight domains, probing five socio-cognitive adaptation axes, with topic-localized scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05563","pdf":"https://arxiv.org/pdf/2606.05563","project":null,"code":"https://github.com/DISL-Lab/SoCRATES","data":null,"hfPaper":"https://huggingface.co/papers/2606.05563"},"evidence":{"snippet":"We introduce SoCRATES, a benchmark for evaluating proactive LLM mediators in realistic, multi-domain testbeds.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":56,"hfDailySubmittedAt":"2026-06-08T00:00:00.000Z","githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05563"},"ranking":{"90d":{"score":31,"rank":226,"coverage":0.7,"confidence":"Medium"}},"description":"SoCRATES is a benchmark for evaluating proactive LLM mediators in realistic, multi-domain testbeds. It contains 600 conflict scenarios across eight domains, probing five socio-cognitive adaptation axes, with topic-localized scoring.","whyItMatters":"Mediation evaluation requires realistic trajectories and topic-specific scoring. SoCRATES provides a structured testbed with socio-cognitive variations, enabling reliable comparison of LLM mediators.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"57fa78e57b3da75f3b3f92c36fd4c142a2ffdddf0fbba27fb1d3582d54bda95e"},"motivation":"Evaluating LLM mediators remains challenging, as mediation unfolds as a real-time trajectory shaped by disputants' shifting emotions, intentions, and context.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05563","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"DISL-Lab","organizationType":"academic-lab","sourceUrl":"https://github.com/DISL-Lab/SoCRATES","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_socsci-repro-bench_8a1a65b5","familyId":"bmf_3424da46c0af","name":"SocSci-Repro-Bench","oneLine":"SocSci-Repro-Bench evaluates AI coding agents on reproducing social science findings from 221 tasks across four disciplines and 13 domains, using studies with known reproducibility outcomes.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11447","pdf":"https://arxiv.org/pdf/2606.11447","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.11447"},"evidence":{"snippet":"Here we introduce SocSci-Repro-Bench, a benchmark of 221 tasks spanning four disciplines and 13 substantive domains, constructed from studies whose results are either fully reproducible with available materials or demonstrably non-reproducible due to missing data, allowing us to isolate agents' reproduction capacity.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.11447"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SocSci-Repro-Bench evaluates AI coding agents on reproducing social science findings from 221 tasks across four disciplines and 13 domains, using studies with known reproducibility outcomes.","whyItMatters":"SocSci-Repro-Bench addresses the lack of systematic evaluation of AI agents' ability to execute computational workflows, providing a protocol to isolate agent performance from issues in reproduction materials and informing practical use of agents in scientific production.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cead3849ff644a663b207bd72d807853980df1f2c8c6f7f62f7d4e77ab5f3a31"},"motivation":"Recent anecdotal evidence suggests that AI coding agents can reproduce published findings when provided with original data and code; yet systematic evaluation across social sciences remains limited.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11447","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_softvtbench_b731832c","familyId":"bmf_eb5e5d2a1a04","name":"SoftVTBench","oneLine":"SoftVTBench is a safety-aware visuo-tactile benchmark for deformable object manipulation in Isaac Sim, with FEM-simulated objects, multi-view RGB, tactile sensing, proprioception, language instructions, and separate Goal Success and Safety Success metrics.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.04234","pdf":"https://arxiv.org/pdf/2607.04234","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.04234"},"evidence":{"snippet":"We present SoftVTBench, a safety-aware visuo-tactile benchmark for physically constrained deformable object manipulation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.04234"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SoftVTBench is a safety-aware visuo-tactile benchmark for deformable object manipulation in Isaac Sim, with FEM-simulated objects, multi-view RGB, tactile sensing, proprioception, language instructions, and separate Goal Success and Safety Success metrics.","whyItMatters":"Existing manipulation benchmarks focus on task success and overlook physical safety, such as avoiding drops or excessive deformation. SoftVTBench addresses this gap by providing a protocol that evaluates both goal achievement and safety, offering decision value for developing safer robotic manipulation policies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"70ef8c359f39494bdcaf94033c185b8bfae3ec293dab0e09e84d968eac48357c"},"motivation":"Deformable object manipulation poses challenges beyond task completion: successful execution must also maintain safe physical interaction, holding the object stably without slip or drop while avoiding excessive deformation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCVW","evidence":"Early version of SoftVTBench, Accepted by ECCVW","evidenceUrl":"https://arxiv.org/abs/2607.04234","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCVW","reviewStatus":"accepted","decisionRaw":"Early version of SoftVTBench, Accepted by ECCVW","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.04234","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Early version of SoftVTBench, Accepted by ECCVW","level":"author-claim"}]}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_softvtbench-a-deformation-aware-visuo-tact_61935c97","familyId":"bmf_3799d0e658b4","name":"SoftVTBench: A Deformation-Aware Visuo-Tactile Dataset and Benchmark for Deformable-Object Manipulation","oneLine":"SoftVTBench is a visuo-tactile dataset and benchmark for deformable-object manipulation, providing expert demonstrations with synchronized sensor data and finite-element ground truth, with a closed-loop evaluation protocol.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-19","firstSeenAt":"2026-08-20","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.18701","pdf":"https://arxiv.org/pdf/2608.18701","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce SoftVTBench, a visuo-tactile dataset for physical-interaction-aware deformable-object manipulation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence","named entity is a model, method, framework, or dataset used in evaluation"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":13,"hfDailySubmittedAt":"2026-08-20T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.18701"},"ranking":{"30d":{"score":49,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":50,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SoftVTBench is a visuo-tactile dataset and benchmark for deformable-object manipulation, providing expert demonstrations with synchronized sensor data and finite-element ground truth, with a closed-loop evaluation protocol.","whyItMatters":"Most manipulation benchmarks evaluate task success alone, ignoring physical interaction quality. SoftVTBench enables evaluation of deformation-aware success, addressing how policies interact with deformable objects.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ea71c1b8664f5b0041da449ff541a210c76f0af48b8340723f17e962e6e421bd"},"motivation":"Physical interaction quality is central to deformable-object manipulation, yet most benchmarks evaluate task success alone.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-20T09:45:58.504563Z","model":"deepseek-v4-flash","decisionReason":"Named dataset and benchmark with clear evaluation protocol and public data availability, though code is not provided but dataset is accessible."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.18701","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-20T09:44:47.308583Z"},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_solarchain-eval_aac1b95f","familyId":"bmf_b1b8147bf474","name":"SolarChain-Eval","oneLine":"A physics-constrained benchmark for evaluating economic agents in decentralized energy markets, with a Gymnasium-compatible MDP and an LLM-based Planner/Auditor layer. It evaluates market utility, physical safety, slippage, action smoothness, spatial fairness, and auditability.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.08681","pdf":"https://arxiv.org/pdf/2607.08681","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.08681"},"evidence":{"snippet":"Therefore, we propose SolarChain-Eval, a physics-constrained benchmark for evaluating trustworthy economic agents.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.08681"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A physics-constrained benchmark for evaluating economic agents in decentralized energy markets, with a Gymnasium-compatible MDP and an LLM-based Planner/Auditor layer. It evaluates market utility, physical safety, slippage, action smoothness, spatial fairness, and auditability.","whyItMatters":"The benchmark addresses the need to evaluate agent trustworthiness in cyber-physical systems, where agents must balance utility with safety. However, without clear public access to the environment and scoring mechanisms, its standalone value is limited.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8fc582953143a68457dab901b4907c9f19bd313e45e09814a690f57a6be3dd5f"},"motivation":"As agentic AI systems are increasingly applied to cyber-physical environments, their evaluation requires assessment of both task performance and trustworthiness.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.08681","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_soliditybench_53feeae1","familyId":"bmf_84813a78a5db","name":"SolidityBench","oneLine":"SolidityBench contains 5,470 repository-level Solidity smart contracts with natural language descriptions, plus SolidityScore, a semantic metric emphasizing domain-critical constructs. It evaluates code generation models.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation"],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19988","pdf":"https://arxiv.org/pdf/2606.19988","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.19988"},"evidence":{"snippet":"To address this gap, we introduce SolidityBench, a benchmark of 5,470 repository-level Solidity smart contracts paired with natural language descriptions.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19988"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SolidityBench contains 5,470 repository-level Solidity smart contracts with natural language descriptions, plus SolidityScore, a semantic metric emphasizing domain-critical constructs. It evaluates code generation models.","whyItMatters":"Domain-specific code generation lacks benchmarks. SolidityBench provides a resource to measure structural and semantic correctness in high-stakes smart contracts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fa2e6184bc9af99a2cc814b3e181087b12ee77b73b2ffe7df02476ff34de2c3d"},"motivation":"Large Language Models (LLMs) have shown strong capabilities in general-purpose code generation, but their effectiveness in specialized software domains remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19988","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SolidityBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.19988","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_sombench_39ae8a74","familyId":"bmf_06b31d962c2c","name":"SoMBench","oneLine":"SoMBench evaluates social intelligence in large language models across 3 primary dimensions, 17 secondary dimensions, and 71 task paradigms, with 284 shared scenarios and 3,481 expert-verified instances.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.23740","pdf":"https://arxiv.org/pdf/2607.23740","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.23740"},"evidence":{"snippet":"For measurement, we introduce SoMBench, a psychology-grounded benchmark spanning 3 primary dimensions, 17 secondary dimensions, and 71 task paradigms.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23740"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SoMBench evaluates social intelligence in large language models across 3 primary dimensions, 17 secondary dimensions, and 71 task paradigms, with 284 shared scenarios and 3,481 expert-verified instances.","whyItMatters":"SoMBench targets the gap in evaluating LLMs' social intelligence, providing a structured benchmark to measure capabilities that are increasingly important for long-term deployment in human environments.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f4cb66738e796a59c9a6315c9f98b2ba928dc9b04da86d6947be750a2ca34778"},"motivation":"As large language models move from isolated task solving toward long-term service in human environments, they require social intelligence: the ability to infer mental states, track social relations, reason over norms, and adapt behavior under context.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.23740","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_sopribench_12eedd2a","familyId":"bmf_4c1de46427bd","name":"SopriBench","oneLine":"SopriBench evaluates user-level privacy leakage from social media posts across text, images, and metadata in 50 synthetic profiles with 1,569 images, scoring via the Privacy Exposure Score (PES) that weights value granularity by contextual sensitivity.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CR"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06784","pdf":"https://arxiv.org/pdf/2606.06784","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.06784"},"evidence":{"snippet":"To address these gaps, we propose SopriBench, a synthetic benchmark guided by leakage patterns abstracted from a private reference corpus of Rednote and Instagram accounts, covering 50 user profiles and 1,569 images with attributes, contextual sensitivity, granularity, leakage type, inference difficulty, and supporting evidence.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06784"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SopriBench evaluates user-level privacy leakage from social media posts across text, images, and metadata in 50 synthetic profiles with 1,569 images, scoring via the Privacy Exposure Score (PES) that weights value granularity by contextual sensitivity.","whyItMatters":"This benchmark addresses the gap in evaluating cumulative cross-post privacy leakage, providing a metric that captures exposure severity rather than binary accuracy, aiding in assessing real-world privacy risks from social media data.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b263b18199ab3aad17385941fe9d79d17046bd29c40867517d2eddb4e4064005"},"motivation":"Public social media posts can reveal private information through weak cues scattered across text, images, or metadata.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.06784","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_soundnessbench_20ec76aa","familyId":"bmf_479eac59a5d6","name":"SoundnessBench","oneLine":"SoundnessBench evaluates LLMs' ability to judge the methodological soundness of research proposals. It contains 1,099 machine-learning research proposals reconstructed from ICLR submissions, labeled with reviewer soundness sub-scores and audited against source papers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["AI Scientist"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2605.30329","pdf":"https://arxiv.org/pdf/2605.30329","project":"https://hosytuyen.github.io/projects/SoundnessBench","code":"https://github.com/hosytuyen/SoundnessBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.30329"},"evidence":{"snippet":"We introduce SoundnessBench, a curated benchmark of 1,099 machine-learning research proposals reconstructed from ICLR submissions, labeled with reviewer soundness sub-scores, and audited against source papers.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30329"},"ranking":{},"description":"SoundnessBench evaluates LLMs' ability to judge the methodological soundness of research proposals. It contains 1,099 machine-learning research proposals reconstructed from ICLR submissions, labeled with reviewer soundness sub-scores and audited against source papers.","whyItMatters":"Autonomous AI research agents need to evaluate research ideas before committing resources, but existing benchmarks do not test this bottleneck. SoundnessBench provides a standardized test for this capability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4db51bc5b4fcf24b5daf1f72fcf0099f6fb6c16d65f28d8678cfc30898fb70da"},"motivation":"Autonomous AI research agents aim to accelerate scientific discovery by automating the research pipeline, from hypothesis generation to peer review.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30329","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"hosytuyen","organizationType":"academic-lab","sourceUrl":"https://github.com/hosytuyen/SoundnessBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_sovereignnegotiation-bench_d54da144","familyId":"bmf_51f52e6461b2","name":"SovereignNegotiation-Bench","oneLine":"SovereignNegotiation-Bench evaluates personal agents in delegated bargaining scenarios, measuring agreement success alongside user utility, privacy, consent, evidence grounding, concession discipline, escalation, and auditability.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.02814","pdf":"https://arxiv.org/pdf/2607.02814","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.02814"},"evidence":{"snippet":"We introduce SovereignNegotiation-Bench, a trace-level multi-turn benchmark for delegated personal-agent negotiation under private utilities, disclosure constraints, evidence requirements, and institutional asymmetry.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02814"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SovereignNegotiation-Bench evaluates personal agents in delegated bargaining scenarios, measuring agreement success alongside user utility, privacy, consent, evidence grounding, concession discipline, escalation, and auditability.","whyItMatters":"Addresses the gap that existing negotiation benchmarks focus on agreement or surplus, potentially overlooking critical user protections in delegated bargaining. Provides a framework for evaluating both strategic and sovereign aspects of personal agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fe8aefa3ff3e5922523bdf85f62552dfb9c97e438c8650dee517b131314b9553"},"motivation":"Personal agents will increasingly negotiate on behalf of users: splitting costs with other personal agents, appealing platform decisions, escalating support disputes, requesting refunds, changing subscriptions, and negotiating deadlines or reimbursements.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02814","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_sovereignpa-bench_ec65a032","familyId":"bmf_dd949d2b2f94","name":"SovereignPA-Bench","oneLine":"SovereignPA-Bench evaluates user-owned personal agents on 120 sovereignty stress scenarios, measuring task success, alignment, privacy, consent, evidence, manipulation, burden, and auditability in evolving intent and platform mediation contexts.","area":"Language & Knowledge","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.05363","pdf":"https://arxiv.org/pdf/2607.05363","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.05363"},"evidence":{"snippet":"We introduce SovereignPA-Bench, an executable benchmark for evaluating user-owned personal agents under evolving intent, platform mediation, privacy boundaries, consent constraints, evidence requirements, and burden tradeoffs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.05363"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SovereignPA-Bench evaluates user-owned personal agents on 120 sovereignty stress scenarios, measuring task success, alignment, privacy, consent, evidence, manipulation, burden, and auditability in evolving intent and platform mediation contexts.","whyItMatters":"Personal agents must balance task completion with user sovereignty. This benchmark quantifies trade-offs across privacy, consent, and manipulation, enabling principled agent design.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d779f2c46df511ec2e6c89eb090de98c1e40fd1096ba1a6774c6ab14a6bb7b3d"},"motivation":"Personal agents are becoming persistent user-owned intermediaries: they remember preferences, filter platform-mediated information, use tools, and negotiate with services.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.05363","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_sp-bench_0f201b2f","familyId":"bmf_bba6eaed485b","name":"SP-Bench","oneLine":"SP-Mind is an autonomous AI agent for spatial proteomics analysis, and SP-Bench is introduced for evaluation with 102 tasks across 18 categories.","area":"Agents & Tool Use","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Agents","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24235","pdf":"https://arxiv.org/pdf/2606.24235","project":null,"code":"https://github.com/tomtommyyuan/spmind","data":null,"hfPaper":"https://huggingface.co/papers/2606.24235"},"evidence":{"snippet":"To rigorously evaluate its capabilities, we introduce SP-Bench, a comprehensive benchmark spanning diverse tissue types, comprising 102 tasks across 18 distinct categories.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":277,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24235"},"ranking":{"90d":{"score":57,"rank":28,"coverage":0.7,"confidence":"Medium"}},"description":"SP-Mind is an autonomous AI agent for spatial proteomics analysis, and SP-Bench is introduced for evaluation with 102 tasks across 18 categories.","whyItMatters":"The benchmark appears tied to the agent's capabilities and lacks a standalone comparison path for other teams.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b88f8a0cb8a58596d67c4a325a6559ac049f720c984c14b1f3d1996fdae72aa3"},"motivation":"Spatial proteomics enables single-cell-resolution characterization of protein expression within tissue architecture, playing a critical role in understanding tumor microenvironments and guiding precision medicine.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026","evidence":"24 pages, 6 figures. Accepted to ICML 2026. Equal contribution by Yucheng Yuan and Yuanfeng Ji","evidenceUrl":"https://arxiv.org/abs/2606.24235","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026","reviewStatus":"accepted","decisionRaw":"24 pages, 6 figures. Accepted to ICML 2026. Equal contribution by Yucheng Yuan and Yuanfeng Ji","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.24235","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"24 pages, 6 figures. Accepted to ICML 2026. Equal contribution by Yucheng Yuan and Yuanfeng Ji","level":"author-claim"}]}],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_sp-transientbench_52ec4ce5","familyId":"bmf_92037a40599b","name":"SP-TransientBench","oneLine":"SP-TransientBench (STB) is a real-captured multi-task benchmark for single-photon perception. It comprises 10 diverse scenes and 10,297 views captured with a solid-state single-photon LiDAR at 256×192 resolution, providing full time-of-flight histograms with multi-return behavior, calibrated camera poses, and standardized metadata. The benchmark evaluates depth estimation, multi-view reconstruction, and 3D semantic understanding, with 13-class semantic annotations for selected scenes. Dedicated data splits and evaluation protocols are provided for each task.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18952","pdf":"https://arxiv.org/pdf/2606.18952","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18952"},"evidence":{"snippet":"To bridge this gap, we introduce SP-TransientBench (STB), a real-captured multi-task benchmark for single photon perception.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18952"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SP-TransientBench (STB) is a real-captured multi-task benchmark for single-photon perception. It comprises 10 diverse scenes and 10,297 views captured with a solid-state single-photon LiDAR at 256×192 resolution, providing full time-of-flight histograms with multi-return behavior, calibrated camera poses, and standardized metadata. The benchmark evaluates depth estimation, multi-view reconstruction, and 3D semantic understanding, with 13-class semantic annotations for selected scenes. Dedicated data splits and evaluation protocols are provided for each task.","whyItMatters":"Existing single-photon perception studies rely on simulated or small-scale captures, lacking a systematic real-world evaluation. STB provides a consistent and reproducible benchmark across multiple 3D vision tasks, enabling comparative assessment of algorithms in photon-starved scenarios and advancing active 3D perception research.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b81b7540def7c0fdfd886aa4d61bcc43435ace241fc354e6e8280ecccc8fc2ea"},"motivation":"Single-photon LiDAR (SPL) based on single-photon avalanche diode (SPAD) sensing enables time-resolved photon measurements with extreme sensitivity, offering unique potential for active 3D perception in photon-starved scenarios.However, real-world single photon perception remains fundamentally challenging due to unique measurement noise and complex multi-return transient phenomena, which jointly complicate geometric…","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18952","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_spade-bench_b8164972","familyId":"bmf_96fc99f9be33","name":"SPADE-Bench","oneLine":"Benchmark for evaluating spontaneous plan-action divergence in agents, integrating actual tool execution and controlled pressure scenarios to distinguish strategic deception from hallucination.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.02380","pdf":"https://arxiv.org/pdf/2606.02380","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02380"},"evidence":{"snippet":"To assess this, we introduce SPADE-Bench, a benchmark designed to evaluate spontaneous plan-action divergence.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02380"},"ranking":{},"description":"Benchmark for evaluating spontaneous plan-action divergence in agents, integrating actual tool execution and controlled pressure scenarios to distinguish strategic deception from hallucination.","whyItMatters":"Addresses the critical risk of agent deception in autonomous systems, but lacks a public comparison path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3fbc72802ddfa61546f9fcfd3295107dc956b1ef8e312a1502d72f7d7a6ec9b9"},"motivation":"As LLM-based agents expand their operational scope, reliability becomes a prerequisite for real-world deployment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02380","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_spapath-bench_cda2e02f","familyId":"bmf_3601f520ca7d","name":"SpaPath-Bench","oneLine":"SpaPath-Bench evaluates pathology foundation models on spatial domain identification using paired whole slide images and spatial transcriptomics data from 42 public slides, measuring partition quality via unsupervised spatial coherence, transcriptomics-referenced agreement, and expert-referenced agreement.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.25764","pdf":"https://arxiv.org/pdf/2605.25764","project":"https://bokai-zhao.github.io/SpaPath-benchboard/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.25764"},"evidence":{"snippet":"We present SpaPath-Bench, a representation level benchmark designed to diagnose spatial representation capability in PFMs.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25764"},"ranking":{},"description":"SpaPath-Bench evaluates pathology foundation models on spatial domain identification using paired whole slide images and spatial transcriptomics data from 42 public slides, measuring partition quality via unsupervised spatial coherence, transcriptomics-referenced agreement, and expert-referenced agreement.","whyItMatters":"Standard task-level endpoints obscure what pathology embeddings encode about tissue spatial structure. SpaPath-Bench provides a representation-level diagnostic that isolates spatial understanding, enabling model developers to compare encoders and methods on a fixed protocol and choose architectures suited for spatially aware computational pathology.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3ae6a971492f16861805e83eae3585e3ce883b02ceb4064c6d049eb5c0a403cd"},"motivation":"Pathology foundation models (PFMs) have emerged as a core approach for learning transferable representations from whole slide images (WSIs), and they are typically benchmarked through downstream clinical endpoints.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25764","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SpaPath-Bench team","organizationType":"academic-lab","sourceUrl":"https://bokai-zhao.github.io/SpaPath-benchboard/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_spar-bench_b75502fb","familyId":"bmf_bc002936817b","name":"SPAR-Bench","oneLine":"Evaluates spatial reasoning in medical vision encoders via eight probes over multi-organ abdominal CT covering coordinate localization, relational reasoning, and spatial queries.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":0.6,"links":{"report":"https://arxiv.org/abs/2608.28092","pdf":"https://arxiv.org/pdf/2608.28092","project":"https://spar-bench.github.io","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We construct SPAR-Bench, eight probes over multi-organ abdominal CT that separate coordinate localization, relational reasoning, and spatial queries, and apply them to five architectural configurations and three medical foundation models, frozen and finetuned.","reasonCodes":["exact named benchmark artifact released in abstract","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.28092"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates spatial reasoning in medical vision encoders via eight probes over multi-organ abdominal CT covering coordinate localization, relational reasoning, and spatial queries.","whyItMatters":"Reveals whether medical vision representations support spatial comparisons rather than recalling canonical anatomy, guiding encoder selection and probing methodology.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T04:51:30.440525Z","inputHash":"e31514ef3f0fb8b87d9097f6217413faf10407c4506fb1b310f9429d3c2d7191"},"motivation":"Interpreting a CT scan means comparing structures on either side, judging how far apart organs sit, and knowing where each one belongs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T04:51:30.440525Z","model":"deepseek-v4-pro","decisionReason":"Formally named SPAR-Bench with a defined eight-probe evaluation and announced public code/data path.","canonicalNameSource":"abstract","canonicalNameEvidence":"We construct SPAR-Bench, eight probes over multi-organ abdominal CT that separate coordinate localization, relational reasoning, and spatial queries"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.28092","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T01:03:30.163531Z"},"attentionForecast":{"score":85,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets a core gap in medical AI and includes findings across architectures, plus a clear project site, likely drawing strong interest."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_spatialbench_9dae04ac","familyId":"bmf_0f6a40719bb8","name":"SpatialBench","oneLine":"SpatialBench is a deterministic, density-aware benchmark for spatial foundation models, spanning 19 datasets, 546 scenes, and five spatial domains. It evaluates 41 models across six paradigms on five task suites—depth, camera pose, trajectory, point-cloud reconstruction, and long-sequence streaming—under four input density settings with precomputed and pinned test frames.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27367","pdf":"https://arxiv.org/pdf/2605.27367","project":"https://ropedia.github.io/SpatialBench/","code":"https://github.com/Ropedia/SpatialBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.27367"},"evidence":{"snippet":"To address this gap, we present SpatialBench, a cross-paradigm, domain-diverse benchmark for spatial foundation models with deterministic sampling.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":73,"hfDailySubmittedAt":null,"githubStars":126,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27367"},"ranking":{},"description":"SpatialBench is a deterministic, density-aware benchmark for spatial foundation models, spanning 19 datasets, 546 scenes, and five spatial domains. It evaluates 41 models across six paradigms on five task suites—depth, camera pose, trajectory, point-cloud reconstruction, and long-sequence streaming—under four input density settings with precomputed and pinned test frames.","whyItMatters":"Spatial foundation models are typically evaluated only on domains they were designed for, making cross-domain generalization difficult to assess. SpatialBench provides a controlled protocol with fixed sampling and multiple density settings, enabling a holistic comparison of generalization across viewpoints, scene domains, and hardware constraints.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"505b4cad6d6f880c864a40b68e5d3d888acd56f5bbabfc8c67e7f4c6e3fe42c8"},"motivation":"While spatial foundation models have demonstrated impressive performance on standard datasets, a critical question remains: are they truly all-round players capable of generalizing robustly across diverse downstream tasks, arbitrary viewpoints, shifting scene domains, varying input densities, and specific hardware constraints?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27367","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Ropedia","organizationType":"academic-lab","sourceUrl":"https://github.com/Ropedia/SpatialBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_61558f1a5c46d186","familyId":"catalog_family_61558f1a5c46d186","name":"SpatialBench Verified","oneLine":"Analysis of spatial transcriptomics data across externally validated biological problems.","description":"Analysis of spatial transcriptomics data across externally validated biological problems.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_61558f1a5c46d186"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/spatialbenchverified"}],"catalogSources":[{"catalog":"benchlm","sourceId":"spatialBenchVerified","url":"https://benchlm.ai/benchmarks/spatialbenchverified","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"LatchBio SpatialBench Verified","format":"Task score","tasks":"115 externally validated spatial transcriptomics problems","successorKey":null}],"catalogCategories":["knowledge"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_spatialbench-long_33cb2dcb","familyId":"bmf_0da3e779f3e3","name":"SpatialBench-Long","oneLine":"SpatialBench-Long evaluates AI agents on long-horizon spatial biology tasks, requiring recovery of biological claims from raw data across 24 evaluations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.28065","pdf":"https://arxiv.org/pdf/2605.28065","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28065"},"evidence":{"snippet":"We introduce SpatialBench-Long, a benchmark for long-horizon spatial biology in which agents must recover biological claims from raw or near-raw data and calibrated experimental context without prescribed methods.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28065"},"ranking":{},"description":"SpatialBench-Long evaluates AI agents on long-horizon spatial biology tasks, requiring recovery of biological claims from raw data across 24 evaluations.","whyItMatters":"Tests agents' ability to synthesize scientific conclusions from complex spatial data, but no public data or code release is mentioned.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8f1d8ec13af1d6b9cd390a0525d15956e4c98483bff966102577795ae15c4ec3"},"motivation":"AI agents are increasingly useful for biological data analysis, but existing benchmarks mostly test broad biological knowledge, executable workflows, or localized analysis steps rather than end-to-end scientific reasoning over spatial measurements.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28065","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_spatialgen-bench_3d213770","familyId":"bmf_54461512c988","name":"SpatialGen-Bench","oneLine":"ProVisE is a framework for evaluating image-generation models on spatial benchmarks by converting visual answers into structured predictions. SpatialGen-Bench is a diagnostic dataset of 470 samples across 14 spatial subtasks.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.21072","pdf":"https://arxiv.org/pdf/2607.21072","project":"https://zju-omniai.github.io/ProVisE/","code":"https://github.com/ZJU-OmniAI/ProVisE","data":null,"hfPaper":"https://huggingface.co/papers/2607.21072"},"evidence":{"snippet":"We further introduce SpatialGen-Bench, a curated diagnostic benchmark of 470 samples across 14 spatial subtasks, four capability levels, and diverse answer forms.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":37,"hfDailySubmittedAt":null,"githubStars":23,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.21072"},"ranking":{"90d":{"score":49,"rank":77,"coverage":0.7,"confidence":"Medium"}},"description":"ProVisE is a framework for evaluating image-generation models on spatial benchmarks by converting visual answers into structured predictions. SpatialGen-Bench is a diagnostic dataset of 470 samples across 14 spatial subtasks.","whyItMatters":"Existing spatial benchmarks rely on text or coordinates, limiting image-generation models. ProVisE adapts visual answers to original metrics, enabling comparison. However, unclear if it is a standalone benchmark or a framework.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a7fd92873345ae54aa6f0b132092a7b892c6298b91082aa2bc60b6a21e25fd1f"},"motivation":"Spatial intelligence is essential for agents to move from static semantic understanding toward interacting with the physical world.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.21072","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_spatialuav_6900b974","familyId":"bmf_e9cbb4f3f030","name":"SpatialUAV","oneLine":"Evaluates spatial intelligence in low-altitude UAV scenarios across 14 task types covering semantic discrimination, spatial relations, collaboration, and motion understanding, with 7 input configurations and 9 answer formats.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.27876","pdf":"https://arxiv.org/pdf/2606.27876","project":null,"code":"https://github.com/Hyu-Zhang/SpatialUAV","data":null,"hfPaper":"https://huggingface.co/papers/2606.27876"},"evidence":{"snippet":"To address these gaps, we introduce SpatialUAV, a real low-altitude UAV benchmark comprising 4,331 curated instances across 14 fine-grained task types, covering semantic discrimination, spatial relation, aerial--aerial collaboration, aerial--ground collaboration, and motion understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.27876"},"ranking":{"90d":{"score":28,"rank":251,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates spatial intelligence in low-altitude UAV scenarios across 14 task types covering semantic discrimination, spatial relations, collaboration, and motion understanding, with 7 input configurations and 9 answer formats.","whyItMatters":"Targets under-evaluated 3D spatial and multi-view reasoning in UAV benchmarks, revealing bottlenecks in cross-view association and geometric reasoning for current VLMs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"de5b56c74e975bed6465829bbf0b94bf4b993a6dc6e85d808af5a576fe45b083"},"motivation":"Spatial intelligence is essential for low-altitude unmanned aerial vehicle (UAV) perception, collaboration, and navigation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.27876","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Hyu-Zhang","organizationType":"academic-lab","sourceUrl":"https://github.com/Hyu-Zhang/SpatialUAV","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_spatialworld_4f21023e","familyId":"bmf_b45842946201","name":"SpatialWorld","oneLine":"SpatialWorld evaluates multimodal agents on interactive spatial reasoning in 760 real-world tasks across eight simulation backends. Agents operate under vision-only partial observability, using a unified text-based action interface. Performance is measured via terminal-state verifiers for task success rate and step efficiency.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09669","pdf":"https://arxiv.org/pdf/2606.09669","project":null,"code":"https://github.com/Hongcheng-Gao/SpatialWorld","data":null,"hfPaper":"https://huggingface.co/papers/2606.09669"},"evidence":{"snippet":"We introduce SpatialWorld, a unified benchmark designed specifically for evaluating the interactive spatial understanding of multimodal agents in complex real-world tasks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":48,"hfDailySubmittedAt":"2026-06-09T00:00:00.000Z","githubStars":54,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09669"},"ranking":{"90d":{"score":55,"rank":34,"coverage":0.7,"confidence":"Medium"}},"description":"SpatialWorld evaluates multimodal agents on interactive spatial reasoning in 760 real-world tasks across eight simulation backends. Agents operate under vision-only partial observability, using a unified text-based action interface. Performance is measured via terminal-state verifiers for task success rate and step efficiency.","whyItMatters":"Existing benchmarks rely on passive VQA or simulator-specific pipelines, failing to assess interactive spatial understanding. SpatialWorld provides a unified, simulator-agnostic protocol with human-validated evaluation, revealing that current models achieve low success rates, highlighting gaps in active exploration and long-horizon planning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0d7ddb4d37209814fd6ecfb0cd6b5bbf2aa8c346230263f46848ee81b8fe5600"},"motivation":"Spatial reasoning is a foundational capability for multimodal large language models (MLLMs) to perceive and operate within the physical world.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09669","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SpatialWorld Team","organizationType":"academic-lab","sourceUrl":"https://github.com/Hongcheng-Gao/SpatialWorld","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_spearbench_7a039de0","familyId":"bmf_e27542ecd67d","name":"SPEARBench","oneLine":"SPEARBench evaluates naturalness in streaming speech-to-speech language models via question-answer interactions, measuring latency, interruptions, speech quality, ASR robustness, language consistency, emotional naturalness, and interpersonal stance.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.05365","pdf":"https://arxiv.org/pdf/2607.05365","project":"https://thomasthebaud.github.io/SPEAR-benchmark-website/#welcome","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.05365"},"evidence":{"snippet":"We introduce SPEARBench, a benchmark for evaluating naturalness in speech-to-speech language models from question-answer interactions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.05365"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SPEARBench evaluates naturalness in streaming speech-to-speech language models via question-answer interactions, measuring latency, interruptions, speech quality, ASR robustness, language consistency, emotional naturalness, and interpersonal stance.","whyItMatters":"Standard speech metrics miss conversational naturalness. This benchmark provides a multidimensional protocol to assess human-like behavior in spoken interactions, critical for user acceptance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fe69f7bf05f0f643db34786aa8abe7971a6dd7f342d4f0dafcde617207ad320f"},"motivation":"Streaming speech-to-speech language models aim to answer spoken queries directly with synthetic speech.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.05365","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_speechdx_58f3f420","familyId":"bmf_89f2b89c21d5","name":"SpeechDx","oneLine":"SpeechDx evaluates clinical speech AI across 12 datasets and 27 tasks covering various conditions, structured by speech production stages. Scoring includes classification accuracy and zero-shot transfer performance.","area":"Speech & Audio","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.17339","pdf":"https://arxiv.org/pdf/2606.17339","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.17339"},"evidence":{"snippet":"We introduce SpeechDx, a large-scale benchmark for clinical speech AI spanning 12 datasets and 27 tasks across diverse health conditions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.17339"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SpeechDx evaluates clinical speech AI across 12 datasets and 27 tasks covering various conditions, structured by speech production stages. Scoring includes classification accuracy and zero-shot transfer performance.","whyItMatters":"Clinical speech AI lacks comparable evaluation across conditions. SpeechDx provides a shared framework to assess generalization and representation quality, guiding progress toward general-purpose models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"30c3333abd92784b111d6a5e52d7e8d60f57e64abe28e8e7f357519f7fb90be4"},"motivation":"Speech offers a uniquely informative window into health by simultaneously engaging neurological, motor, respiratory, and vocal systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.17339","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_9bfa0b50a90e6699","familyId":"catalog_family_9bfa0b50a90e6699","name":"Spider","oneLine":"A large-scale, complex and cross-domain semantic parsing and text-to-SQL dataset annotated by 11 college students. Contains 10,181 questions and 5,693 unique complex SQL queries on 200 databases with multiple tables, covering 138 different domains. Requires models to generalize to both new SQL queries and new database schemas, making it distinct from previous semantic parsing tasks that use single databases.","description":"A large-scale, complex and cross-domain semantic parsing and text-to-SQL dataset annotated by 11 college students. Contains 10,181 questions and 5,693 unique complex SQL queries on 200 databases with multiple tables, covering 138 different domains. Requires models to generalize to both new SQL queries and new database schemas, making it distinct from previous semantic parsing tasks that use single databases.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/spider","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_9bfa0b50a90e6699"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/spider"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"spider","url":"https://llm-stats.com/benchmarks/spider","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_spider-2-0-aifunc_d9873e24","familyId":"bmf_020b339c7771","name":"Spider 2.0-AIFunc","oneLine":"Extends text-to-SQL to AI-native SQL workflows with 465 instances across 125 real-world databases on Snowflake. Tasks require using AI functions like classification and sentiment analysis. Evaluates execution accuracy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.06229","pdf":"https://arxiv.org/pdf/2607.06229","project":null,"code":"https://github.com/Leolty/Spider2-AIFunc","data":null,"hfPaper":"https://huggingface.co/papers/2607.06229"},"evidence":{"snippet":"We introduce Spider 2.0-AIFunc, a benchmark of 465 verified instances across 125 real-world databases covering six types of AI functions on the Snowflake platform.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06229"},"ranking":{"90d":{"score":23,"rank":305,"coverage":0.7,"confidence":"Medium"}},"description":"Extends text-to-SQL to AI-native SQL workflows with 465 instances across 125 real-world databases on Snowflake. Tasks require using AI functions like classification and sentiment analysis. Evaluates execution accuracy.","whyItMatters":"Addresses the gap of benchmarks not covering AI-native SQL capabilities that are increasingly available in cloud platforms. Provides a reusable dataset and evaluation harness for a new task type.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b4de75b68993207d60c9728ecee5c188aba765d4980f4e90b8594f79cfab9fa0"},"motivation":"Major cloud data platforms now expose large language model capabilities as native SQL functions, enabling analysts to perform classification, filtering, sentiment analysis, extraction, similarity search, and aggregation within ordinary SQL queries.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06229","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Leolty","organizationType":"community","sourceUrl":"https://github.com/Leolty/Spider2-AIFunc","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_f2b646370579c919","familyId":"catalog_family_f2b646370579c919","name":"Spider 2.0-Lite","oneLine":"A text-to-SQL benchmark over realistic warehouse-scale schemas, reported by Interfaze for model comparison.","description":"A text-to-SQL benchmark over realistic warehouse-scale schemas, reported by Interfaze for model comparison.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://github.com/xlang-ai/Spider2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f2b646370579c919"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/spider2lite"}],"catalogSources":[{"catalog":"benchlm","sourceId":"spider2Lite","url":"https://benchlm.ai/benchmarks/spider2lite","paperUrl":"https://github.com/xlang-ai/Spider2","year":"2024","fullName":"Spider 2.0-Lite","format":"Execution accuracy","tasks":"Text-to-SQL queries","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_spieval_a21bc309","familyId":"bmf_8905e75a39cb","name":"SPIEval","oneLine":"SPIEval is a human-curated benchmark for evaluating large language models as mobile assistants that retrieve and reason over personal information scattered across multiple apps. It comprises 250 tasks across five cognitive capabilities, 4,335 fictional personal records in 10 simulated apps, and supports multi-turn interaction through 21 tools. Models receive underspecified user instructions and must search records and invoke tools; final execution calls are compared to human-annotated gold calls at the parameter level.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.10692","pdf":"https://arxiv.org/pdf/2608.10692","project":null,"code":null,"data":"https://huggingface.co/datasets/Junjie-Ye/SPIEval","hfPaper":"https://huggingface.co/papers/2608.10692"},"evidence":{"snippet":"To address this gap, we introduce SPIEval, a human-curated benchmark grounded in five cognitive capabilities (i.e., reasoning, disambiguation, integration, preference inference, and multi-intent decomposition).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":12,"hfDailySubmittedAt":"2026-08-12T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":335,"hfDatasetLikes":1},"source":{"type":"arxiv","id":"2608.10692"},"ranking":{"30d":{"score":50,"rank":28,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":11,"datasetRankPopulation":30},"90d":{"score":49,"rank":73,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":26,"datasetRankPopulation":66}},"description":"SPIEval is a human-curated benchmark for evaluating large language models as mobile assistants that retrieve and reason over personal information scattered across multiple apps. It comprises 250 tasks across five cognitive capabilities, 4,335 fictional personal records in 10 simulated apps, and supports multi-turn interaction through 21 tools. Models receive underspecified user instructions and must search records and invoke tools; final execution calls are compared to human-annotated gold calls at the parameter level.","whyItMatters":"Existing benchmarks do not target the challenge of leveraging scattered personal information across apps in mobile assistant settings. SPIEval provides a controlled, verifiable evaluation environment grounded in five cognitive capabilities, enabling assessment of model capabilities in realistic mobile contexts. The reported results show substantial room for improvement and highlight fundamental limitations in information localization and search efficiency, offering practical guidance for deploying LLM-based assistants.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"edf6c3676d14db9361b58c1f98994c02d946f5484df4db8cf02829bc5ba991c2"},"motivation":"Large language models (LLMs) are increasingly deployed as mobile assistants, where a key challenge is leveraging personal information scattered across multiple applications (apps) to complete user instructions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10692","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_spike-bench_3b80754c","familyId":"bmf_71b5d4d6ae47","name":"SPIKE-Bench","oneLine":"SPIKE-Bench evaluates LLM biosecurity risks using 631 curated toxin-design prompts and a three-stage funnel (compliance, plausibility, predicted toxicity) producing the Functional Harmfulness Rate.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Safety"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Runnable","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02684","pdf":"https://arxiv.org/pdf/2608.02684","project":null,"code":"https://github.com/PKU-Alignment/SPIKE-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2608.02684"},"evidence":{"snippet":"To address this evaluation blind spot, we introduce SPIKE-Bench, coupling 631 curated toxin-design prompts across seven functional categories with the SPIKE funnel, a three-stage protocol that filters output through compliance, biological plausibility, and predicted toxicity, producing stage-level diagnostics and an aggregate function-aware metric: the Functional Harmfulness Rate (FHR).","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02684"},"ranking":{"30d":{"score":31,"rank":78,"coverage":0.55,"confidence":"Low"},"90d":{"score":31,"rank":233,"coverage":0.55,"confidence":"Low"}},"description":"SPIKE-Bench evaluates LLM biosecurity risks using 631 curated toxin-design prompts and a three-stage funnel (compliance, plausibility, predicted toxicity) producing the Functional Harmfulness Rate.","whyItMatters":"Provides a function-aware metric for biosecurity evaluation beyond refusal rates, enabling comparison of models on predicted functional harm.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f5a2312fc8ce8ce06a37a24ed20b5009d784e622db5fd68cc48c0356f8266367"},"motivation":"Large Language Models (LLMs) are accelerating biological research, yet this same capability poses a critical biosecurity threat: models that assist in protein engineering can equally be prompted to generate predicted toxin-like sequences, potentially lowering the barrier to biological misuse.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"COLM 2026","evidence":"Accepted to COLM 2026. 40 pages, 9 figures","evidenceUrl":"https://arxiv.org/abs/2608.02684","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"COLM 2026","reviewStatus":"accepted","decisionRaw":"Accepted to COLM 2026. 40 pages, 9 figures","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.02684","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to COLM 2026. 40 pages, 9 figures","level":"author-claim"}]}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_sportmv-bench_a6dbc37e","familyId":"bmf_90df5079f183","name":"SportMV-Bench","oneLine":"SportMV-Bench evaluates multi-view sports video understanding with 1022 multi-view bundles and 3015 QA pairs across 10 sports, covering perception, rule-aware event interpretation, and adjudicative reasoning.","area":"Vision & 3D","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.11844","pdf":"https://arxiv.org/pdf/2607.11844","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.11844"},"evidence":{"snippet":"To address this gap, we introduce SportMV-Bench, a comprehensive benchmark built from official match recordings, through a dedicated pipeline combining LLM-based generation, MLLM-based verification, and human filtering to ensure quality and consistency.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.11844"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SportMV-Bench evaluates multi-view sports video understanding with 1022 multi-view bundles and 3015 QA pairs across 10 sports, covering perception, rule-aware event interpretation, and adjudicative reasoning.","whyItMatters":"This benchmark addresses the lack of multi-view sports video evaluation, revealing that MLLMs struggle with fine-grained visual perception and view selection, guiding future development in video reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e662c385e93034586c119dd77ed3b05e762da52087f17d1425c4611aa6c1da22"},"motivation":"Recent Multimodal Large Language Models (MLLMs) achieve strong performance on single-view video understanding benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.11844","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_spreadsheetbench_a0b75616","familyId":"bmf_15499180d75b","name":"SpreadsheetBench","oneLine":"SpreadsheetBench 2 evaluates spreadsheet agents on end-to-end business workflows across generation, debugging, and visualization tasks. It includes 321 tasks from authentic business data, with multi-sheet workbooks requiring cross-sheet reasoning. The benchmark provides a unified multi-turn agent scaffold and evaluation scripts for reproducibility.","area":"Code & Software","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29955","pdf":"https://arxiv.org/pdf/2606.29955","project":"https://spreadsheetbench.github.io/","code":"https://github.com/RUCKBReasoning/SpreadsheetBench-2","data":null,"hfPaper":"https://huggingface.co/papers/2606.29955"},"evidence":{"snippet":"We introduce \\textsc{SpreadsheetBench 2}, a workflow-level benchmark for spreadsheet agents that covers three task categories: generation, debugging, and visualization.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":29,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29955"},"ranking":{"90d":{"score":41,"rank":144,"coverage":0.7,"confidence":"Medium"}},"description":"SpreadsheetBench 2 evaluates spreadsheet agents on end-to-end business workflows across generation, debugging, and visualization tasks. It includes 321 tasks from authentic business data, with multi-sheet workbooks requiring cross-sheet reasoning. The benchmark provides a unified multi-turn agent scaffold and evaluation scripts for reproducibility.","whyItMatters":"Existing spreadsheet benchmarks focus on isolated operations, failing to capture real-world workflow complexity. SpreadsheetBench 2 addresses this gap by assessing agents on tasks that require multi-step coordination, cross-sheet reasoning, and deliverable-level outcomes. It provides a challenging testbed for improving reliable spreadsheet automation, with current models achieving only 34.89% overall accuracy.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"329d728aee104cb1c8c7d0c29699f8dd02c46b3c2b39565d5c320a36736ba71b"},"motivation":"Spreadsheets are widely used for business analysis, financial modeling, reporting, and decision-making.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.29955","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"RUCKBReasoning","organizationType":"academic-lab","sourceUrl":"https://github.com/RUCKBReasoning/SpreadsheetBench-2","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"specific"},{"id":"catalog_6735aba5a7a45579","familyId":"catalog_family_6735aba5a7a45579","name":"SpreadsheetBench 2","oneLine":"SpreadsheetBench 2 evaluates office automation agents on spreadsheet analysis, reasoning, and manipulation tasks.","description":"SpreadsheetBench 2 evaluates office automation agents on spreadsheet analysis, reasoning, and manipulation tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agentic","Productivity","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6735aba5a7a45579"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/spreadsheetbench2"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/spreadsheetbench-2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"spreadsheetBench2","url":"https://benchlm.ai/benchmarks/spreadsheetbench2","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"SpreadsheetBench 2","format":"Agent task-completion score","tasks":"Spreadsheet analysis and editing tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"spreadsheetbench-2","url":"https://llm-stats.com/benchmarks/spreadsheetbench-2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","productivity","agents","tool calling"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_5c8580698885b99c","familyId":"catalog_family_5c8580698885b99c","name":"SpreadSheetBench-v1","oneLine":"SpreadSheetBench-v1 evaluates office automation agents on spreadsheet reasoning and manipulation tasks, measuring the ability to analyze, transform, and operate on spreadsheet data through tools.","description":"SpreadSheetBench-v1 evaluates office automation agents on spreadsheet reasoning and manipulation tasks, measuring the ability to analyze, transform, and operate on spreadsheet data through tools.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Productivity","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/spreadsheetbench-v1","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5c8580698885b99c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/spreadsheetbench-v1"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"spreadsheetbench-v1","url":"https://llm-stats.com/benchmarks/spreadsheetbench-v1","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["productivity","agents","tool calling"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_sqbench_9c1d0337","familyId":"bmf_ba6726f3c0ac","name":"SQBench","oneLine":"SQBench v1.0 evaluates language-model agents on 220 production-oriented tasks organized into L1 atomic capabilities, L2 composite skills, and L3 business scenarios. Tasks require processing input assets, using tools, and producing a specified deliverable. Scoring computes Completion, Risk Penalty, and Performance from a 10D Risk Matrix; Strict Pass requires Completion=1 and Risk Penalty=0.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.23123","pdf":"https://arxiv.org/pdf/2607.23123","project":null,"code":"https://github.com/shaqiu-ai/SQBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.23123"},"evidence":{"snippet":"We introduce SQBench, a benchmark for evaluating production-oriented task delivery by language-model agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23123"},"ranking":{"90d":{"score":23,"rank":366,"coverage":0.55,"confidence":"Low"}},"description":"SQBench v1.0 evaluates language-model agents on 220 production-oriented tasks organized into L1 atomic capabilities, L2 composite skills, and L3 business scenarios. Tasks require processing input assets, using tools, and producing a specified deliverable. Scoring computes Completion, Risk Penalty, and Performance from a 10D Risk Matrix; Strict Pass requires Completion=1 and Risk Penalty=0.","whyItMatters":"The benchmark targets delivery under domain constraints, a shared weakness in current models. Its scoring separates functional completion from risk, which could support decisions about agent deployment in production workflows. However, without public access to tasks and full results, its practical value is limited for external comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0a99db05505fe6f210c3dcaf8248dbaff264e6b7172899589d277490580ac512"},"motivation":"Existing evaluations of large language models cover knowledge, reasoning, coding, and tool use, but they rarely treat a verifiable deliverable produced within a constrained workflow as the unit of evaluation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.23123","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_6dd05549572d185a","familyId":"catalog_family_6dd05549572d185a","name":"SQuALITY","oneLine":"SQuALITY (Summarization-format QUestion Answering with Long Input Texts, Yes!) is a long-document summarization dataset built by hiring highly-qualified contractors to read public-domain short stories (3000-6000 words) and write original summaries from scratch. Each document has five summaries: one overview and four question-focused summaries. Designed to address limitations in existing summarization datasets by providing high-quality, faithful summaries.","description":"SQuALITY (Summarization-format QUestion Answering with Long Input Texts, Yes!) is a long-document summarization dataset built by hiring highly-qualified contractors to read public-domain short stories (3000-6000 words) and write original summaries from scratch. Each document has five summaries: one overview and four question-focused summaries. Designed to address limitations in existing summarization datasets by providing high-quality, faithful summaries.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Long Context","Summarization"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/squality","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6dd05549572d185a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/squality"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"squality","url":"https://llm-stats.com/benchmarks/squality","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","long context","summarization"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"bm_sre-bench_d394a59b","familyId":"bmf_d2aa6d139e85","name":"SRE-Bench","oneLine":"SRE-Bench evaluates AI agents on reverse engineering of binaries compiled from 19 private real-world-scale C programs (average 16.9K LoC) with 44 anti-analysis primitives, yielding 262 binary instances and 1572 deterministically graded tasks.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.11469","pdf":"https://arxiv.org/pdf/2608.11469","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.11469"},"evidence":{"snippet":"To this end, we introduce SRE-Bench, the first realistic, contamination-free RE benchmark.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11469"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SRE-Bench evaluates AI agents on reverse engineering of binaries compiled from 19 private real-world-scale C programs (average 16.9K LoC) with 44 anti-analysis primitives, yielding 262 binary instances and 1572 deterministically graded tasks.","whyItMatters":"Existing benchmarks for agentic cybersecurity miss either contamination control or realistic scale. SRE-Bench addresses this gap by combining private, real-world-scale binaries with deterministic grading, enabling reliable measurement of agent performance in binary analysis and highlighting the gap between source-code and binary security capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2aeed24a8279cc24f3368bd4c0dda9ef029042a24563a915d52b0627e3d81ba2"},"motivation":"AI agents are rapidly improving in cybersecurity capabilities when the source code is available for analysis, yet much of the software most consequential to cybersecurity, including malware, firmware, and proprietary applications, is available only as binaries.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11469","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_ssmnbench_555dd856","familyId":"bmf_9224c6cbf216","name":"SSMNBench","oneLine":"SSMNBench is a diagnostic benchmark for cross-view human and human-object understanding, comprising 3,300 QA pairs categorized into Single-View Sufficiency (SVS) and Multi-View Necessity (MVN) tasks. It evaluates multimodal LLMs by perturbing view availability to assess distraction robustness and cross-view evidence integration.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.25634","pdf":"https://arxiv.org/pdf/2606.25634","project":null,"code":"https://github.com/gtc-gh/SSMNBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.25634"},"evidence":{"snippet":"To address this issue, we introduce SSMNBench, a diagnostic benchmark comprising 3,300 curated QA pairs for cross-view human and human-object understanding.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.25634"},"ranking":{"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SSMNBench is a diagnostic benchmark for cross-view human and human-object understanding, comprising 3,300 QA pairs categorized into Single-View Sufficiency (SVS) and Multi-View Necessity (MVN) tasks. It evaluates multimodal LLMs by perturbing view availability to assess distraction robustness and cross-view evidence integration.","whyItMatters":"Addresses the gap in evaluating genuine cross-view synthesis versus reliance on single-image semantics in MLLMs. It provides a rigorous framework for diagnosing limitations in cross-view understanding, guiding development of multimodal architectures for complex scenes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4a9df5d8069cd7572a4b03dc3a5607df0f30dcb4599eb3349614eab789a59594"},"motivation":"Multimodal Large Language Models (MLLMs) have shown remarkable progress in single-image perception, yet their ability to reason about complex cross-view human-centric scenes remains largely unverified.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.25634","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_st-bench_eeee9e2f","familyId":"bmf_c12a5ba8d0fc","name":"ST-Bench","oneLine":"ST-Bench is a verification benchmark for evaluating certified robustness of spatio-temporal neural networks on autonomous driving (Udacity) and activity recognition (UCF-101) tasks, using spatio-temporal perturbation constraints.","area":"Safety & Trustworthiness","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Automotive"],"capabilities":["Robustness"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.09746","pdf":"https://arxiv.org/pdf/2606.09746","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.09746"},"evidence":{"snippet":"To spur further progress in this field, we propose ST-Bench, a verification benchmark for autonomous driving and activity recognition, to systematically evaluate verifiable robustness.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09746"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ST-Bench is a verification benchmark for evaluating certified robustness of spatio-temporal neural networks on autonomous driving (Udacity) and activity recognition (UCF-101) tasks, using spatio-temporal perturbation constraints.","whyItMatters":"Existing robustness verification methods rely on overly conservative assumptions or are computationally prohibitive. ST-Bench provides a realistic, constrained perturbation model for video inputs, enabling tighter approximations and more meaningful robustness comparisons for safety-critical applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"85a0aa69d53bb7351576bac24f6aca41b9443a1ff5d32b6f96ce3155c4c63d0c"},"motivation":"With AI increasingly deployed in safety-critical systems, providing formal robustness guarantees for the underlying models is essential.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"9th International Symposium on AI Verification (SAIV 2026)","evidence":"Accepted at the 9th International Symposium on AI Verification (SAIV 2026)","evidenceUrl":"https://arxiv.org/abs/2606.09746","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"9th International Symposium on AI Verification (SAIV 2026)","reviewStatus":"accepted","decisionRaw":"Accepted at the 9th International Symposium on AI Verification (SAIV 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.09746","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at the 9th International Symposium on AI Verification (SAIV 2026)","level":"author-claim"}]}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_stabilitybench_c271b529","familyId":"bmf_56ebe7cdf845","name":"StabilityBench","oneLine":"StabilityBench is a benchmark operator that converts single-turn benchmark queries into multi-turn interaction histories with injected user simulations, evaluating LLM performance stability across demographic proxies and sycophantic baits on existing benchmarks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20558","pdf":"https://arxiv.org/pdf/2607.20558","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20558"},"evidence":{"snippet":"We propose StabilityBench, a principled, general and model-agnostic benchmark operator that transforms single-turn benchmark queries into multi-turn interaction histories.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20558"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"StabilityBench is a benchmark operator that converts single-turn benchmark queries into multi-turn interaction histories with injected user simulations, evaluating LLM performance stability across demographic proxies and sycophantic baits on existing benchmarks.","whyItMatters":"Static benchmarks may not capture real-world conversational variability; StabilityBench highlights performance instability under realistic multi-turn conditions, motivating more realistic evaluation settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0b80505a27112ab586568270cfdea41ea236860ba0814c7d7f75c69bb8f682c0"},"motivation":"AI Assistants are increasingly deployed in high-stakes settings, such as healthcare or government services.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20558","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_stage-claw_61b30974","familyId":"bmf_002eae0d56e7","name":"STAGE-Claw","oneLine":"STAGE-Claw is an automated framework for building and evaluating personal-agent tasks in state-based computing environments, with a benchmark of 40 tasks. Evaluations measure final system state correctness.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.10394","pdf":"https://arxiv.org/pdf/2606.10394","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.10394"},"evidence":{"snippet":"Given a task hint, STAGE-Claw automatically creates and validates a realistic benchmark task with its environment, task prompts, ground truth, and related verification programs.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.10394"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"STAGE-Claw is an automated framework for building and evaluating personal-agent tasks in state-based computing environments, with a benchmark of 40 tasks. Evaluations measure final system state correctness.","whyItMatters":"The framework addresses the need for scalable and realistic evaluation of personal agents, moving beyond sandboxed and static tasks to state-based verification. This supports progress in agent reliability and practical deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"54d9942e93beef802a2cd9ebc1975ee378cf0b8dc7519b187f6a0826ebbafb06"},"motivation":"Large language models are increasingly used to power personal agents for everyday applications, but evaluating these agents remains a challenge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.10394","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_stakebench_be4543bc","familyId":"bmf_bb61545ac7a2","name":"StakeBench","oneLine":"StakeBench is a stakeholder-centric benchmark for prompt-injection attacks in web agents for online shopping. It decomposes risk into 12 attack objectives across three stakeholder classes, with 264 adversarial cases across 12 product categories.","area":"Language & Knowledge","applicationDomains":["Cybersecurity","Finance & Economics"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity","Financial Services"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13385","pdf":"https://arxiv.org/pdf/2606.13385","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.13385"},"evidence":{"snippet":"To capture these properties, we introduce StakeBench, a stakeholder-centric benchmark that systematically categorizes and attributes harm in real-world web agent systems for online shopping.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13385"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"StakeBench is a stakeholder-centric benchmark for prompt-injection attacks in web agents for online shopping. It decomposes risk into 12 attack objectives across three stakeholder classes, with 264 adversarial cases across 12 product categories.","whyItMatters":"Prompt-injection risk is victim-dependent. StakeBench captures asymmetric consequences for different stakeholders, which is overlooked by attack-centric evaluations.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ce4c91daad8cf70bdc63538671da0ef1118b36aa9229478867c967e28a5752e2"},"motivation":"LLM-based web agents are increasingly deployed in real-world settings such as e-commerce, where they interact extensively with untrusted web content while executing actions that carry direct financial consequences.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13385","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"bm_staminabench_cb49af23","familyId":"bmf_5030babf657a","name":"StaminaBench","oneLine":"StaminaBench stress-tests coding agents over 100 interaction turns of change requests, measuring how many consecutive turns they handle before failing, with programmatically generated tasks and black-box HTTP evaluation.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19613","pdf":"https://arxiv.org/pdf/2606.19613","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.19613"},"evidence":{"snippet":"We introduce StaminaBench, a benchmark that measures the stamina of coding agents: how many consecutive interaction turns (change requests) they can handle before failing.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19613"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"StaminaBench stress-tests coding agents over 100 interaction turns of change requests, measuring how many consecutive turns they handle before failing, with programmatically generated tasks and black-box HTTP evaluation.","whyItMatters":"Fills the gap in evaluating multi-turn coding agent stamina, which is critical for real-world vibe-coding sessions that often extend over many turns.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d57ee1987a79485a6b3445097afcaa1f466de6d28e5d360fb520bf31eddc373a"},"motivation":"We introduce StaminaBench, a benchmark that measures the stamina of coding agents: how many consecutive interaction turns (change requests) they can handle before failing.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19613","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_stancebench_98c43cc6","familyId":"bmf_8befc508f2a5","name":"StanceBench","oneLine":"StanceBench evaluates interpersonal stance in conversational speech across 9 dimensions using LLM-as-a-judge on the Seamless Interaction corpus.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.22658","pdf":"https://arxiv.org/pdf/2607.22658","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.22658"},"evidence":{"snippet":"We introduce StanceBench, a benchmark for measuring interpersonal stance in conversational speech and evaluating audio-capable LLMs as automated judges.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.22658"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"StanceBench evaluates interpersonal stance in conversational speech across 9 dimensions using LLM-as-a-judge on the Seamless Interaction corpus.","whyItMatters":"Addresses the gap in benchmarks for prosody and interactional nuance in speech-to-speech models, focusing on judge bias and robustness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e307dc5489a394a269ec99c7fe7f269e93a311562eb3aec2ba23c4f7274e622c"},"motivation":"Speech-to-speech dialogue models increasingly depend on prosody and interactional nuance to convey social intent, yet benchmarks for these cues remain limited.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"Interspeech 2026","evidence":"Accepted to Interspeech 2026","evidenceUrl":"https://arxiv.org/abs/2607.22658","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"Interspeech 2026","reviewStatus":"accepted","decisionRaw":"Accepted to Interspeech 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.22658","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to Interspeech 2026","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_stanceflip_415aae19","familyId":"bmf_e22e85b4e626","name":"StanceFlip","oneLine":"StanceFlip is a benchmark for multimodal conversational stance flipping forecasting with two subtasks: sextuple extraction and flip attribution.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.24191","pdf":"https://arxiv.org/pdf/2607.24191","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.24191"},"evidence":{"snippet":"To address these limitations, we propose StanceFlip, a benchmark designed for multimodal conversational stance flipping forecasting over multi-turn dialogues across five modalities and multi-scenarios, which includes two novel subtasks: 1) Multimodal Stance Sextuple Extraction, extracting holder, target, emotion, sentiment, stance, and rationale as static state snapshots of dialogue to capture fine-grained cognitive structures.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.24191"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"StanceFlip is a benchmark for multimodal conversational stance flipping forecasting with two subtasks: sextuple extraction and flip attribution.","whyItMatters":"It addresses gaps in dynamic stance evolution and multimodal cues, but lacks a public reuse path and stable scoring contract.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8e511dd5b77afe9d288726c2d0bd2353e829077cef005ae11e334becb7007a0e"},"motivation":"Conversational stance detection has shifted from static text analysis to dynamic multimodal modeling.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.24191","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_startupbench_c9f7419c","familyId":"bmf_064eca785918","name":"StartupBench","oneLine":"Evaluates whether general-purpose agents can complete market-validated, end-to-end professional workflows and deliver usable work products.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics","Science & Research"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":["Long-horizon task execution","Tool use","Complex instruction following","Professional artifact generation"],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.17800","pdf":"https://arxiv.org/pdf/2608.17800","project":"https://startupbench.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.17800"},"evidence":{"snippet":"We introduce \\textbf{StartupBench}, an E2E agent benchmark grounded in market-validated AI startup products.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":9,"hfDailySubmittedAt":"2026-08-19T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.17800"},"ranking":{"30d":{"score":47,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":49,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"StartupBench is a benchmark for general-purpose agents on end-to-end workflows derived from market-validated AI startup products, with deliverable-oriented tasks and fine-grained rubrics.","whyItMatters":"Existing agent benchmarks are researcher-selected, leaving real-world task performance uncertain. StartupBench measures agents on tasks with demonstrated demand.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"36c7ed0bbf540064241e50febdc82ed09641f45fd042dd87433ec946514c5077"},"motivation":"Recent advances in Large Language Models(LLMs) and agents have substantially improved the ability of AI systems to execute complex tasks.","constructionDetail":"StartupBench measures whether agents can complete realistic professional workflows and produce usable final work products across six broad disciplines.","detail":{"taskBreakdown":["Medical & Healthcare","Finance","Legal","Business & Management","STEM & Computer Science","Education & Humanities"],"protocol":{"tasks":"97 end-to-end workflow tasks","primaryMetric":"Importance-weighted rubric score (0–100); success rate at score ≥90","version":"arXiv v1"},"leaderboardUrl":"https://startupbench.github.io/#leaderboard"},"curation":{"state":"source-reviewed","reviewedAt":"2026-08-20","sources":["https://arxiv.org/abs/2608.17800","https://arxiv.org/html/2608.17800","https://startupbench.github.io/"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.17800","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"evaluationMode":"score_submission","availability":{"paperStatus":"available","githubStatus":"not_found","hfDatasetStatus":"not_found","evaluatorStatus":"described_not_released","submissionStatus":"not_found"},"capabilityGroups":["Agents","Tool Calling"],"domainScope":"cross-domain"},{"id":"bm_statabench_6a243141","familyId":"bmf_81f1ac51c69f","name":"StatABench","oneLine":"StatABench evaluates LLMs' statistical analysis capabilities through two components: Stat-Closed, 404 questions across 18 topics in multiple formats, and Stat-Open, 30 complex modeling tasks from competitions, scored via LLM-as-judge.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22977","pdf":"https://arxiv.org/pdf/2606.22977","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.22977"},"evidence":{"snippet":"To bridge this gap, we introduce StatABench (Statistical AnalysisBenchmark), a benchmark designed to systematically assess LLMs' statistical analysis capabilities.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22977"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"StatABench evaluates LLMs' statistical analysis capabilities through two components: Stat-Closed, 404 questions across 18 topics in multiple formats, and Stat-Open, 30 complex modeling tasks from competitions, scored via LLM-as-judge.","whyItMatters":"Addresses the need for systematic evaluation of LLMs in statistical analysis, revealing performance gaps and challenges in tool-grounded reasoning and modeling.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ace7c1855e953724bb15c82ed04e7a3c984e30bfcf14f2807724e35d5f62534f"},"motivation":"Statistical analysis is a broad, complex field requiring both domain knowledge and tool proficiency.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22977","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_statmechbench-v0_d1047ef0","familyId":"bmf_d86596b80f05","name":"StatMechBench-v0","oneLine":"A benchmark of six Ising-type problems for evaluating LLM agents' ability to discover statistical mechanical mappings from raw partition functions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.26367","pdf":"https://arxiv.org/pdf/2607.26367","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.26367"},"evidence":{"snippet":"To probe this question, we introduce StatMechBench-v0, a benchmark of six Ising-type problems covering transfer-matrix methods, gauge-removable disorder, and planar/Pfaffian structure.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26367"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark of six Ising-type problems for evaluating LLM agents' ability to discover statistical mechanical mappings from raw partition functions.","whyItMatters":"This probes AI capabilities for structural discovery in theoretical physics, highlighting limitations in current LLM reasoning and the need for verification beyond numerical agreement.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ef72598a7d2b5379357e9526d30e81fee1c3b988fbaa7039df7bf54ef8526974"},"motivation":"An important skill in theoretical physics is to recognize when a new problem can be transformed into a known model.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"AID-Wild workshop at CAIS 2026","evidence":"Accepted to the AID-Wild workshop at CAIS 2026","evidenceUrl":"https://arxiv.org/abs/2607.26367","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"AID-Wild workshop at CAIS 2026","reviewStatus":"accepted","decisionRaw":"Accepted to the AID-Wild workshop at CAIS 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.26367","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to the AID-Wild workshop at CAIS 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_stealthbench_8fce6971","familyId":"bmf_94101338dd62","name":"StealthBench","oneLine":"StealthBench measures operational stealth of autonomous offensive-security agents across six OPSEC dimensions, using 14 dockerized task scenarios and a three-model judge panel, with metrics like safe success rate and Stealth@Solve.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.26314","pdf":"https://arxiv.org/pdf/2607.26314","project":"https://stealthbench.com","code":"https://github.com/GangGreenTemperTatum/stealthbench","data":"https://huggingface.co/datasets/0xmoose/stealthbench","hfPaper":"https://huggingface.co/papers/2607.26314"},"evidence":{"snippet":"We introduce StealthBench,a benchmark that measures operational stealth in autonomous offensive-security agents across six operational security (OPSEC) dimensions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":"2026-07-30T00:00:00.000Z","githubStars":7,"githubScope":"benchmark_repo","hfDatasetDownloads":29,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2607.26314"},"ranking":{"90d":{"score":29,"rank":250,"coverage":1.0,"confidence":"High","datasetDownloadRank":66,"datasetRankPopulation":66}},"description":"StealthBench measures operational stealth of autonomous offensive-security agents across six OPSEC dimensions, using 14 dockerized task scenarios and a three-model judge panel, with metrics like safe success rate and Stealth@Solve.","whyItMatters":"It addresses the gap where agents find vulnerabilities but fail tradecraft, systematic across models. The benchmark supports development of stealth-aware agents and automated OPSEC monitoring for safe deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f6e481bfa5ae085eb48f1356775981d044114bdd028b8c553f7f5c4ae84fdc04"},"motivation":"Stealth, the discipline of achieving an objective without revealing your presence, capabilities, or collected intelligence, is what separates sophisticated operators from detectable ones.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26314","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"stealthbench.com","organizationType":"benchmark-organization","sourceUrl":"https://stealthbench.com","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_steelbench_6bf9f31b","familyId":"bmf_c0bc763f6367","name":"SteelBench","oneLine":"STEELBENCH evaluates vision-language models on per-worker activity recognition and safety-rule reasoning in industrial CCTV footage. It includes 1,345 clips with dense annotations and a provenance-aware audit protocol.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Safety","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.05264","pdf":"https://arxiv.org/pdf/2607.05264","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.05264"},"evidence":{"snippet":"We introduce STEELBENCH, a diagnostic benchmark for industrial surveillance that jointly evaluates per-worker activity recognition, safety-rule reasoning, and annotation provenance.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.05264"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"STEELBENCH evaluates vision-language models on per-worker activity recognition and safety-rule reasoning in industrial CCTV footage. It includes 1,345 clips with dense annotations and a provenance-aware audit protocol.","whyItMatters":"Industrial surveillance poses unique visual and procedural challenges. This benchmark reveals large gaps in VLM performance and shows how annotation provenance can inflate accuracy, guiding reliable deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b95fe3291dce8586dd5090da725478438b20b1e36996f31f3e4c6369a81b8646"},"motivation":"Existing video benchmarks evaluate action recognition on consumer videos, egocentric recordings, or simulated industrial environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.05264","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_steerbench-work_754206df","familyId":"bmf_9f160b21637e","name":"SteerBench-Work","oneLine":"SteerBench-Work evaluates agent steering decisions at action boundaries in workplace scenarios across seven domains, with 106 incident-anchored scenarios and evidence-reversed mirrors, scored on correct proceed/hold boundaries.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences","Finance & Economics"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech","Financial Services"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12654","pdf":"https://arxiv.org/pdf/2608.12654","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12654"},"evidence":{"snippet":"We introduce SteerBench-Work, an incident-anchored, bidirectional benchmark for that decision in workplace agents across developer operations, customer service, finance, legal, medical, HR, and security.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12654"},"ranking":{"30d":{"score":39,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SteerBench-Work evaluates agent steering decisions at action boundaries in workplace scenarios across seven domains, with 106 incident-anchored scenarios and evidence-reversed mirrors, scored on correct proceed/hold boundaries.","whyItMatters":"Addresses the critical pre-commit decision in long-running agents, where a single step can have significant consequences. The benchmark reveals that models tend to over-refuse authorized actions, providing valuable insight for calibrating agent behavior to avoid both unsafe actions and unnecessary delays.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c5c73171cb06508b48d45dd868634ea611d199ecd1014db18559877225383f61"},"motivation":"Long-running LLM agents act through tools, and a single step can send an email, merge a pull request, or wire a payment.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12654","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"catalog_919a11f948095f31","familyId":"catalog_family_919a11f948095f31","name":"STEM","oneLine":"A comprehensive multimodal benchmark dataset with 448 skills and 1,073,146 questions spanning all STEM subjects (Science, Technology, Engineering, Mathematics), designed to test neural models' vision-language STEM skills based on K-12 curriculum. Unlike existing datasets that focus on expert-level ability, this dataset includes fundamental skills designed around educational standards.","description":"A comprehensive multimodal benchmark dataset with 448 skills and 1,073,146 questions spanning all STEM subjects (Science, Technology, Engineering, Mathematics), designed to test neural models' vision-language STEM skills based on K-12 curriculum. Unlike existing datasets that focus on expert-level ability, this dataset includes fundamental skills designed around educational standards.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/stem","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_919a11f948095f31"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/stem"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"stem","url":"https://llm-stats.com/benchmarks/stem","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","multimodal","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_stembind_bdce22aa","familyId":"bmf_140387d0197e","name":"StemBind","oneLine":"StemBind is a diagnostic benchmark with shared-stem questions to attribute failures in abstract visual reasoning. It includes perception, rule, and full tasks with stage annotations but no official code or dataset release.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.00148","pdf":"https://arxiv.org/pdf/2606.00148","project":"https://hexixiang.github.io/StemBind","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.00148"},"evidence":{"snippet":"We introduce StemBind, a shared-stem diagnostic benchmark that probes the same visual stem with three aligned questions: Perception (what is in the image), Rule (what pattern governs it), and Full (which option completes it), so a final-answer error can be attributed to a specific sub-step on the same evidence.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00148"},"ranking":{},"description":"StemBind is a diagnostic benchmark with shared-stem questions to attribute failures in abstract visual reasoning. It includes perception, rule, and full tasks with stage annotations but no official code or dataset release.","whyItMatters":"It aims to localize reasoning failures to specific sub-steps, potentially guiding improvements in multimodal models, but lacks a public path for independent verification.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"36262cddaf9bdc858bb3fda3ebf8046138fb7d7e81302c6a1639a6ade468db6e"},"motivation":"Multimodal large language models (MLLMs) often know the rule but pick the wrong answer: on abstract visual reasoning (AVR) tasks, a model can describe what it sees and name the underlying pattern, yet still fail to choose the matching candidate.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00148","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_stemgym_d150c717","familyId":"bmf_14f397cd3d3e","name":"STEMGym","oneLine":"STEMGym is an open-source Gymnasium benchmark of 15 physics-simulated STEM worlds across five materials and four characterization tasks, scored by DEC-AUC.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29592","pdf":"https://arxiv.org/pdf/2606.29592","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.29592"},"evidence":{"snippet":"We introduce STEMGym, an open-source Gymnasium benchmark of 15 physics-simulated STEM worlds spanning five materials, three difficulty levels, and four characterisation tasks, scored by the Dose-Efficiency Curve area (DEC-AUC), a single scalar capturing the information-vs-dose Pareto frontier.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29592"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"STEMGym is an open-source Gymnasium benchmark of 15 physics-simulated STEM worlds across five materials and four characterization tasks, scored by DEC-AUC.","whyItMatters":"Addresses the need to evaluate perception, navigation, and planning trade-offs in autonomous electron microscopy under dose budgets.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a638a75b98ea1f772fe83faaaa67121f945ac07dd317a823db3270f6bc953412"},"motivation":"A central premise of autonomous scientific imaging is that smarter navigation, whether Bayesian, RL-based, or otherwise adaptive, is the principal lever for sample-efficient acquisition.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.29592","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_stepjack_2287a7d0","familyId":"bmf_43b09f4a2eb3","name":"StepJack","oneLine":"StepJack evaluates computer-use agents against multi-step indirect prompt injection attacks, where adversarial goals are decomposed into innocuous sub-steps across a chain of pages. It comprises 480 test examples across platforms and instruction types, with attack success rate as the primary metric.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.06477","pdf":"https://arxiv.org/pdf/2608.06477","project":null,"code":"https://github.com/BorealisAI/StepJack","data":null,"hfPaper":"https://huggingface.co/papers/2608.06477"},"evidence":{"snippet":"With this pipeline, we build StepJack, a CUA safety benchmark with 480 test examples.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06477"},"ranking":{"30d":{"score":34,"rank":68,"coverage":0.55,"confidence":"Low"},"90d":{"score":33,"rank":216,"coverage":0.55,"confidence":"Low"}},"description":"StepJack evaluates computer-use agents against multi-step indirect prompt injection attacks, where adversarial goals are decomposed into innocuous sub-steps across a chain of pages. It comprises 480 test examples across platforms and instruction types, with attack success rate as the primary metric.","whyItMatters":"The benchmark addresses the evaluation gap in agent safety against sophisticated, staged prompt injection attacks that previous single-step benchmarks fail to capture, offering a standardized way to assess and compare defenses for computer-use agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"125383de0677db78ed7bbb12a15bc852e83fc2c788fd04f0283a6d6b919c4feb"},"motivation":"Computer-use agents (CUAs) face a growing threat from indirect prompt injection, where adversarial instructions are planted in the environment such as web pages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06477","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"BorealisAI","organizationType":"company-research-lab","sourceUrl":"https://github.com/BorealisAI/StepJack","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_stereogenbench_78fe1ed3","familyId":"bmf_a246b7b4c19d","name":"StereoGenBench","oneLine":"StereoGenBench is a synthetic multi-camera benchmark for stereo generation, rendered in Unreal Engine with a rigid six-camera array. It provides calibrated view pairs across baseline regimes, along with RGB, metric depth, intrinsics, and poses for each scene.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-05-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.23237","pdf":"https://arxiv.org/pdf/2605.23237","project":null,"code":null,"data":"https://huggingface.co/datasets/stereo-dataset/stereo-dataset","hfPaper":"https://huggingface.co/papers/2605.23237"},"evidence":{"snippet":"We introduce StereoGenBench, a synthetic Unreal Engine benchmark designed to make baseline-regime sensitivity and target-camera consistency measurable under matched scene content.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":1604,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2605.23237"},"ranking":{},"description":"StereoGenBench is a synthetic multi-camera benchmark for stereo generation, rendered in Unreal Engine with a rigid six-camera array. It provides calibrated view pairs across baseline regimes, along with RGB, metric depth, intrinsics, and poses for each scene.","whyItMatters":"Stereo generation and view synthesis require controlled baseline and intrinsics for evaluation, which existing resources lack. This benchmark enables measuring sensitivity to baseline regimes and target-camera consistency under matched scene content, supporting development of stereo generation models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b61dfbd5cb24164ed4560890f3148878bf3ab598d1ae9ce73b14f8edaa79b960"},"motivation":"Stereo image and video generation, stereo geometry estimation, and condition-controlled view synthesis require paired data in which the variables that determine binocular geometry -- camera baseline, intrinsics, scene depth, and camera motion -- are known and controllable.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.23237","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_stocbench_915b6314","familyId":"bmf_d3fa20132772","name":"StocBench","oneLine":"StocBench evaluates generative models for probabilistic forecasting of stochastic fluid flows, using a two-dimensional Kolmogorov flow with stochastic forcing, measuring one-step distributional accuracy and preservation of the enstrophy spectrum during autoregressive rollouts.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-23","firstSeenAt":"2026-08-25","recognitionConfidence":0.75,"links":{"report":"http://arxiv.org/abs/2608.22309v1","pdf":"https://arxiv.org/pdf/2608.22309v1","project":null,"code":"https://github.com/tum-pbs/stocbench","data":null,"hfPaper":null},"evidence":{"snippet":"We benchmark transport-based generative models as well as distillation-based few-step methods for the probabilistic forecasting of stochastic fluid flows, with a particular focus on performance under limited inference budgets.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22309"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"StocBench evaluates generative models for probabilistic forecasting of stochastic fluid flows, using a two-dimensional Kolmogorov flow with stochastic forcing, measuring one-step distributional accuracy and preservation of the enstrophy spectrum during autoregressive rollouts.","whyItMatters":"It provides a controlled stochastic dynamics testbed to compare transport-based and distillation-based generative models under limited inference budgets, distinguishing aleatoric and epistemic uncertainty through a deterministic control variant.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"8ca0d279867ce67b0422ca0f793bbe95a78bcb85da44370055b1e9b49f474cdb"},"motivation":"We benchmark transport-based generative models as well as distillation-based few-step methods for the probabilistic forecasting of stochastic fluid flows, with a particular focus on performance under limited inference budgets.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally named, provides a defined evaluation object (stochastic Kolmogorov flow) with repeatable scoring (distributional accuracy, enstrophy spectrum), and offers a public code repository for reuse, meeting publication criteria.","canonicalNameSource":"paper_title","canonicalNameEvidence":"StocBench: A Benchmark for Generative Modeling of Stochastic Dynamics"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22309v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":45,"confidence":"Medium","horizon":"7d","reason":"The niche focus on stochastic fluid dynamics benchmarks and moderate breadth suggest modest initial attention, though the explicit benchmark framing and code release may support engagement."},"evaluationMode":"public_reusable","publishers":[{"name":"Technical University of Munich, Physics-Based Simulation Group","organizationType":"academic-lab","sourceUrl":"https://github.com/tum-pbs/stocbench","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_storylensbench_8f70dc12","familyId":"bmf_de15c99faa91","name":"STORYLENSBENCH","oneLine":"STORYLENSBENCH benchmarks preference-aligned story rewriting with structured story books and reader profiles, plus reward model and rewriting model.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.28073","pdf":"https://arxiv.org/pdf/2605.28073","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28073"},"evidence":{"snippet":"Motivated by this, we introduce STORYLENSBENCH, a large-scale benchmark for preference-aligned story rewriting, comprising structured story books, multi-dimensional reader preference profiles, and ranked context-aware rewritten stories.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28073"},"ranking":{},"description":"STORYLENSBENCH benchmarks preference-aligned story rewriting with structured story books and reader profiles, plus reward model and rewriting model.","whyItMatters":"Focuses on context-aware narrative enrichment, but the benchmark is not publicly released; only the models are proposed.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"77833d44ed605c5d6a0ab1a802b2f1b90b9af11487fc6d42d8d212ea5705ad9c"},"motivation":"Story rewriting aims to adapt existing narratives to diverse reader preferences while preserving plot consistency and narrative coherence.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28073","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_strad_bb4cb514","familyId":"bmf_48621e764b76","name":"StrAD","oneLine":"StrAD is a benchmark for long-form audio description generation on full-length videos across diverse genres, reformulating the task as streaming dense video captioning without ground-truth timestamps.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.12549","pdf":"https://arxiv.org/pdf/2608.12549","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.12549"},"evidence":{"snippet":"We introduce StrAD, a benchmark for long-form AD generation on full-length videos spanning diverse genres such as movies, documentaries, short films, performances, and video games.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.12549"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"StrAD is a benchmark for long-form audio description generation on full-length videos across diverse genres, reformulating the task as streaming dense video captioning without ground-truth timestamps.","whyItMatters":"Addresses the lack of evaluation for full-video AD generation, providing a measurable benchmark that includes both fine-tuned and zero-shot approaches, which is essential for scaling accessibility.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5f039f2caf09668bca00d8ee9a5aa4e283525258425625bb09bec64edbe4e345"},"motivation":"Visual content is the dominant medium of communication, yet without audio descriptions (ADs), it remains inaccessible to blind and low-vision people.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12549","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_strategybench_2da1f1e5","familyId":"bmf_0e668381666c","name":"StrategyBench","oneLine":"StrategyBench is a benchmark for evaluating explicit strategy induction in large language models. It selects strategy-inducible tasks from BIG-Bench, constructs reference strategies, and defines evaluation metrics along two dimensions: strategy quality and downstream utility.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.23475v1","pdf":"https://arxiv.org/pdf/2608.23475v1","project":"https://anonymous.4open.science/r/StrategyBench-D53C","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To evaluate this ability, we propose StrategyBench, which selects strategy-inducible tasks from BIG-Bench, constructs reference strategies, and defines evaluation metrics along two dimensions: strategy quality and downstream utility.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23475"},"ranking":{"today":{"score":48,"rank":9,"coverage":0.4,"confidence":"Low"},"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"StrategyBench is a benchmark for evaluating explicit strategy induction in large language models. It selects strategy-inducible tasks from BIG-Bench, constructs reference strategies, and defines evaluation metrics along two dimensions: strategy quality and downstream utility.","whyItMatters":"The benchmark addresses the gap in evaluating whether LLMs can explicitly abstract task rules from examples, which is important for adapting to data-scarce scenarios. It provides a systematic analysis of strategy induction across task variations and model configurations.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"76a93fcd83bd226f8d82ee00ed199e8830a3feb3ddf690dc89866d53cc102165"},"motivation":"As large language models are increasingly used in data-scarce and evolving task scenarios, few-shot in-context learning (ICL) has become a key paradigm for task adaptation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with defined metrics and a public project link.","canonicalNameSource":"abstract","canonicalNameEvidence":"we propose StrategyBench, which selects strategy-inducible tasks from BIG-Bench"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.23475v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":50,"confidence":"Medium","horizon":"7d","reason":"The benchmark tackles a core capability in LLM adaptation and provides a rigorous evaluation, likely to gain traction in AI research communities."},"evaluationMode":"score_submission","publishers":[{"name":"StrategyBench Team","organizationType":"academic-lab","sourceUrl":"https://anonymous.4open.science/r/StrategyBench-D53C","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_streamarena_4967785a","familyId":"bmf_5aa1f09351b6","name":"StreamArena","oneLine":"StreamArena evaluates hour-scale streaming video understanding across 243 full-length videos (avg 88.8 min) with 3,646 open-ended QA pairs, covering real-time perception, historical retrospection, proactive interaction, and multimodal tool use. Includes a standardized runner and LLM-as-judge scorer.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.05703","pdf":"https://arxiv.org/pdf/2608.05703","project":null,"code":"https://github.com/JIA-Lab-research/StreamArena","data":null,"hfPaper":"https://huggingface.co/papers/2608.05703"},"evidence":{"snippet":"We introduce StreamArena, a benchmark for hour-scale, interactive streaming video understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":16,"hfDailySubmittedAt":"2026-08-10T00:00:00.000Z","githubStars":32,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.05703"},"ranking":{"30d":{"score":51,"rank":23,"coverage":0.85,"confidence":"High"},"90d":{"score":49,"rank":74,"coverage":0.7,"confidence":"Medium"}},"description":"StreamArena evaluates hour-scale streaming video understanding across 243 full-length videos (avg 88.8 min) with 3,646 open-ended QA pairs, covering real-time perception, historical retrospection, proactive interaction, and multimodal tool use. Includes a standardized runner and LLM-as-judge scorer.","whyItMatters":"Addresses the lack of benchmarks for long-horizon, interactive streaming video understanding, where short clips and multiple-choice formats allow shortcuts. Provides a rigorous, open-ended evaluation to assess progress in continuous, interactive multimodal agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"671f0a7dc9f6685e9fefd6ba2ee599a227ef4daf4d22af3efd839b23f53bcd68"},"motivation":"Deploying autonomous multimodal agents in continuous, real-world environments requires them to ingest unbounded audio-visual streams and maintain hour-scale memory.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.05703","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"JIA-Lab-research","organizationType":"academic-lab","sourceUrl":"https://github.com/JIA-Lab-research/StreamArena","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_streammembench_6345b534","familyId":"bmf_9ab5cd1646d9","name":"StreamMemBench","oneLine":"StreamMemBench is a streaming benchmark for evaluating agent memory with two-step task sequences around evidence anchors from EgoLife egocentric streams, with four diagnostic metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.14571","pdf":"https://arxiv.org/pdf/2606.14571","project":null,"code":"https://github.com/landian60/StreamMemBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.14571"},"evidence":{"snippet":"We introduce StreamMemBench, a streaming benchmark that constructs a two-step task sequence around each evidence anchor from EgoLife egocentric streams.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":24,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.14571"},"ranking":{"90d":{"score":39,"rank":160,"coverage":0.7,"confidence":"Medium"}},"description":"StreamMemBench is a streaming benchmark for evaluating agent memory with two-step task sequences around evidence anchors from EgoLife egocentric streams, with four diagnostic metrics.","whyItMatters":"Existing memory benchmarks test recall or task improvement in isolation. StreamMemBench evaluates the trajectory from streaming observations to future-oriented assistance, diagnosing whether memory systems use evidence and incorporate feedback.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1e9604ff519227981ed24535395a4b69998f542eff33a3c1122d816caba19716"},"motivation":"A central role of personal-agent memory is to turn stored information and prior interactions into future-oriented assistance.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"Findings of EMNLP 2026","evidence":"Accepted to Findings of EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2606.14571","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"Findings of EMNLP 2026","reviewStatus":"accepted","decisionRaw":"Accepted to Findings of EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.14571","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to Findings of EMNLP 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_streamprofilebench_ccd7b568","familyId":"bmf_5cbc6af978f5","name":"StreamProfileBench","oneLine":"StreamProfileBench evaluates LLMs on fine-grained streaming user profiling. Models maintain a rolling persona summary from a stream of user posts and predict which tags from a candidate pool the user will engage with next. Includes over 120,000 posts from 7,000+ users across five Chinese platforms with metrics for recall, novelty, stability, and error rates.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.25758","pdf":"https://arxiv.org/pdf/2605.25758","project":null,"code":"https://github.com/WaterWang-001/StreamProfileBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.25758"},"evidence":{"snippet":"To bridge this gap, we introduce StreamProfileBench, a large-scale benchmark for fine-grained streaming user profiling.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25758"},"ranking":{},"description":"StreamProfileBench evaluates LLMs on fine-grained streaming user profiling. Models maintain a rolling persona summary from a stream of user posts and predict which tags from a candidate pool the user will engage with next. Includes over 120,000 posts from 7,000+ users across five Chinese platforms with metrics for recall, novelty, stability, and error rates.","whyItMatters":"Existing user profiling benchmarks use static data, failing to capture real-world streaming UGC and rapidly evolving interests. StreamProfileBench provides a dynamic evaluation that measures plasticity-stability balance, revealing conservative bias in LLMs and enabling practical improvements for personalized systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bb535c7a9f14f0b08f2e3098bc7a1a486aa3380373a7bad8c12425f61406dbd5"},"motivation":"Large Language Models (LLMs) have reshaped user profiling, yet current evaluations mainly focus on static data snapshots.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25758","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"StreamProfileBench Team","organizationType":"community","sourceUrl":"https://github.com/WaterWang-001/StreamProfileBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_stylisticbias_ee73496d","familyId":"bmf_110092e1dedb","name":"StylisticBias","oneLine":"StylisticBias is a benchmark of 25,000 images with controlled single-attribute variations to evaluate attribute-level social bias in multimodal LLMs. It measures how visual cues shift model judgments in 25 scenarios.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.20527","pdf":"https://arxiv.org/pdf/2606.20527","project":"https://hf.co/datasets/shaghayegh/stylistic-bias-dataset","code":"https://github.com/timo-cavelius/StylisticBias","data":null,"hfPaper":"https://huggingface.co/papers/2606.20527"},"evidence":{"snippet":"We introduce StylisticBias, a controlled benchmark for evaluating attribute-level social bias in MLLMs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":"2026-06-22T00:00:00.000Z","githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.20527"},"ranking":{"90d":{"score":27,"rank":289,"coverage":0.7,"confidence":"Medium"}},"description":"StylisticBias is a benchmark of 25,000 images with controlled single-attribute variations to evaluate attribute-level social bias in multimodal LLMs. It measures how visual cues shift model judgments in 25 scenarios.","whyItMatters":"Social bias in multimodal models is often confounded by identity. StylisticBias isolates visual cues, enabling targeted bias diagnosis and mitigation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ff40a8290eef1406bddf205706c66c56314b768d849515374ac802451f91aad1"},"motivation":"Multimodal large language models (MLLMs) are increasingly deployed in personally and societally consequential settings, yet the visual cues that shape how these models judge people remain poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"non-archival workshops AI4Good and Culture x AI at ICML 2026","evidence":"Accepted to the non-archival workshops AI4Good and Culture x AI at ICML 2026","evidenceUrl":"https://arxiv.org/abs/2606.20527","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"non-archival workshops AI4Good and Culture x AI at ICML 2026","reviewStatus":"accepted","decisionRaw":"Accepted to the non-archival workshops AI4Good and Culture x AI at ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.20527","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to the non-archival workshops AI4Good and Culture x AI at ICML 2026","level":"author-claim"}]}],"publishers":[{"name":"StylisticBias Team","organizationType":"academic-lab","sourceUrl":"https://github.com/timo-cavelius/StylisticBias","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_subtlememory_d858f4b9","familyId":"bmf_1e2fb9ee2660","name":"SubtleMemory","oneLine":"SubtleMemory evaluates fine-grained relational memory discrimination in long-horizon AI agents. It contains 1,522 evaluation instances over 10 long histories, grounded in 1,090 relation-controlled memory-variant sets, spanning user-related and non-user-related queries.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05761","pdf":"https://arxiv.org/pdf/2606.05761","project":null,"code":"https://github.com/Yummytanmo/SubtleMemory","data":null,"hfPaper":"https://huggingface.co/papers/2606.05761"},"evidence":{"snippet":"To address this gap, we introduce SubtleMemory, a benchmark for fine-grained relational memory discrimination in long-running AI agents.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":19,"hfDailySubmittedAt":"2026-06-08T00:00:00.000Z","githubStars":22,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05761"},"ranking":{"90d":{"score":46,"rank":90,"coverage":0.7,"confidence":"Medium"}},"description":"SubtleMemory evaluates fine-grained relational memory discrimination in long-horizon AI agents. It contains 1,522 evaluation instances over 10 long histories, grounded in 1,090 relation-controlled memory-variant sets, spanning user-related and non-user-related queries.","whyItMatters":"Existing long-term memory benchmarks rarely probe how agents preserve and use relations among memories during downstream tasks. SubtleMemory provides a relation-controlled evaluation to measure this capability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c60f79137f82b8c4cb4a9959e7c1b3b6742fc91694d53915bf360fb02c37b4a8"},"motivation":"Persistent AI assistants, such as OpenClaw, accumulate large collections of related memories over long-term interactions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05761","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SubtleMemory Project","organizationType":"community","sourceUrl":"https://github.com/Yummytanmo/SubtleMemory","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_suichat-cn_1a4b7900","familyId":"bmf_8261cf7b37ba","name":"SuiChat-CN","oneLine":"SuiChat-CN is a Chinese group-chat benchmark for contextual suicide risk assessment, with 13,312 segments from 1,406 users.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27911","pdf":"https://arxiv.org/pdf/2605.27911","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.27911"},"evidence":{"snippet":"We introduce SuiChat-CN, a Chinese group-chat benchmark for contextual suicide risk assessment.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27911"},"ranking":{},"description":"SuiChat-CN is a Chinese group-chat benchmark for contextual suicide risk assessment, with 13,312 segments from 1,406 users.","whyItMatters":"Pioneers risk assessment in group chats but the dataset is restricted for ethical reasons, limiting its public use.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"88488da6ef2109333436f60cf7881eae7dfad1584f6bd4bae1ce2a7d77feba7a"},"motivation":"Suicide is a critical global public health challenge, causing approximately 720,000 deaths each year and calling for timely, effective prevention strategies.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27911","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_54a2a373859605d8","familyId":"catalog_family_54a2a373859605d8","name":"SummScreenFD","oneLine":"SummScreenFD is the ForeverDreaming subset of the SummScreen dataset for abstractive screenplay summarization, comprising pairs of TV series transcripts and human-written recaps from 88 different shows. The dataset provides a challenging testbed for abstractive summarization where plot details are often expressed indirectly in character dialogues and scattered across the entirety of the transcript, requiring models to find and integrate these details to form succinct plot descriptions.","description":"SummScreenFD is the ForeverDreaming subset of the SummScreen dataset for abstractive screenplay summarization, comprising pairs of TV series transcripts and human-written recaps from 88 different shows. The dataset provides a challenging testbed for abstractive summarization where plot details are often expressed indirectly in character dialogues and scattered across the entirety of the transcript, requiring models to find and integrate these details to form succinct plot descriptions.","area":"Long Context","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context","Summarization"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/summscreenfd","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_54a2a373859605d8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/summscreenfd"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"summscreenfd","url":"https://llm-stats.com/benchmarks/summscreenfd","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["long context","summarization"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Long Context & Memory"],"domainScope":"general"},{"id":"catalog_40e1ff1283bd4d04","familyId":"catalog_family_40e1ff1283bd4d04","name":"SUNRGBD","oneLine":"SUNRGBD evaluates RGB-D scene understanding and 3D grounding capabilities.","description":"SUNRGBD evaluates RGB-D scene understanding and 3D grounding capabilities.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Spatial Reasoning","3D","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/sunrgbd","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_40e1ff1283bd4d04"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/sunrgbd"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"sunrgbd","url":"https://llm-stats.com/benchmarks/sunrgbd","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["spatial reasoning","3d","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_7d7e9a75ac4d072e","familyId":"catalog_family_7d7e9a75ac4d072e","name":"SuperChem","oneLine":"SuperChem is a benchmark of advanced chemistry problems requiring expert-level domain knowledge and reasoning.","description":"SuperChem is a benchmark of advanced chemistry problems requiring expert-level domain knowledge and reasoning.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","Chemistry"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/superchem","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7d7e9a75ac4d072e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/superchem"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"superchem","url":"https://llm-stats.com/benchmarks/superchem","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","chemistry"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_56ba0cb792f6bfce","familyId":"catalog_family_56ba0cb792f6bfce","name":"SuperGLUE","oneLine":"SuperGLUE is a new benchmark styled after GLUE with a new set of more difficult language understanding tasks, improved resources, and a new public leaderboard. It includes 8 primary tasks: BoolQ (Boolean Questions), CB (CommitmentBank), COPA (Choice of Plausible Alternatives), MultiRC (Multi-Sentence Reading Comprehension), ReCoRD (Reading Comprehension with Commonsense Reasoning), RTE (Recognizing Textual Entailment), WiC (Word-in-Context), and WSC (Winograd Schema Challenge). The benchmark evaluates diverse language understanding capabilities including reading comprehension, commonsense reasoning, causal reasoning, coreference resolution, textual entailment, and word sense disambiguation across multiple domains.","description":"SuperGLUE is a new benchmark styled after GLUE with a new set of more difficult language understanding tasks, improved resources, and a new public leaderboard. It includes 8 primary tasks: BoolQ (Boolean Questions), CB (CommitmentBank), COPA (Choice of Plausible Alternatives), MultiRC (Multi-Sentence Reading Comprehension), ReCoRD (Reading Comprehension with Commonsense Reasoning), RTE (Recognizing Textual Entailment), WiC (Word-in-Context), and WSC (Winograd Schema Challenge). The benchmark evaluates diverse language understanding capabilities including reading comprehension, commonsense reasoning, causal reasoning, coreference resolution, textual entailment, and word sense disambiguation across multiple domains.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/superglue","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_56ba0cb792f6bfce"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/superglue"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"superglue","url":"https://llm-stats.com/benchmarks/superglue","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning","general"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_6cd5a7b958e562cd","familyId":"catalog_family_6cd5a7b958e562cd","name":"SuperGPQA","oneLine":"SuperGPQA is a comprehensive benchmark that evaluates large language models across 285 graduate-level academic disciplines. The benchmark contains 25,957 questions covering 13 broad disciplinary areas including Engineering, Medicine, Science, and Law, with specialized fields in light industry, agriculture, and service-oriented domains. It employs a Human-LLM collaborative filtering mechanism with over 80 expert annotators to create challenging questions that assess graduate-level knowledge and reasoning capabilities.","description":"SuperGPQA is a comprehensive benchmark that evaluates large language models across 285 graduate-level academic disciplines. The benchmark contains 25,957 questions covering 13 broad disciplinary areas including Engineering, Medicine, Science, and Law, with specialized fields in light industry, agriculture, and service-oriented domains. It employs a Human-LLM collaborative filtering mechanism with over 80 expert annotators to create challenging questions that assess graduate-level knowledge and reasoning capabilities.","area":"Mathematical Reasoning","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Knowledge","Legal","Math","Physics","Reasoning","Finance","General","Healthcare","Chemistry","Economics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2502.14739","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6cd5a7b958e562cd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/supergpqa"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/supergpqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"superGpqa","url":"https://benchlm.ai/benchmarks/supergpqa","paperUrl":"https://arxiv.org/abs/2502.14739","year":"2025","fullName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","format":"Multiple choice questions","tasks":"285 disciplines","successorKey":null},{"catalog":"llm-stats","sourceId":"supergpqa","url":"https://llm-stats.com/benchmarks/supergpqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","legal","math","physics","reasoning","finance","general","healthcare","chemistry","economics"],"catalogModelCount":34,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"bm_surakshaeval_6928de7a","familyId":"bmf_9fb058d84496","name":"SurakshaEval","oneLine":"SurakshaEval evaluates the safety of LLMs across ten major Indian languages and English, using human-written prompts covering generic and region-specific scenarios, with a structured scoring protocol.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07862","pdf":"https://arxiv.org/pdf/2608.07862","project":null,"code":"https://github.com/debobanerjee/SurakshaEval","data":null,"hfPaper":"https://huggingface.co/papers/2608.07862"},"evidence":{"snippet":"To address this gap, we introduce SurakshaEval, a novel safety benchmark composed of human-written prompts spanning real-world scenarios, explicitly designed for ten major Indian languages - Assamese, Bengali, Gujarati, Hindi, Kannada, Malayalam, Marathi, Punjabi, Tamil, and Telugu, along with English.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07862"},"ranking":{"30d":{"score":8,"rank":167,"coverage":0.85,"confidence":"High"},"90d":{"score":15,"rank":399,"coverage":0.7,"confidence":"Medium"}},"description":"SurakshaEval evaluates the safety of LLMs across ten major Indian languages and English, using human-written prompts covering generic and region-specific scenarios, with a structured scoring protocol.","whyItMatters":"Existing safety benchmarks are largely English-centric, leaving a gap for multilingual and culturally grounded safety assessment. SurakshaEval offers a public protocol to benchmark LLM safety in Indian languages, aiding deployment in diverse linguistic contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e1c84909b838af12685cfca2c2dbae3628bfdb6f910f490a3a3bc8da9212d43e"},"motivation":"Existing safety evaluation datasets for large language models (LLMs) predominantly focus on English and Western contexts, often overlooking the linguistic diversity and culturally grounded safety risks present in other languages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07862","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SurakshaEval Benchmark Team","organizationType":"academic-lab","sourceUrl":"https://github.com/debobanerjee/SurakshaEval","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_ce00f2c163a1ff7b","familyId":"catalog_family_ce00f2c163a1ff7b","name":"SURDS","oneLine":"SURDS is a benchmark for spatial understanding and reasoning in autonomous-driving scenes.","description":"SURDS is a benchmark for spatial understanding and reasoning in autonomous-driving scenes.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/surds","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ce00f2c163a1ff7b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/surds"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"surds","url":"https://llm-stats.com/benchmarks/surds","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_surgvla-bench_a2867ef6","familyId":"bmf_71013fdc6c32","name":"SurgVLA-Bench","oneLine":"Evaluates vision-language-action models in laparoscopic surgical robotics across 8 tasks (atomic, conditional, composite) on the SurRoL simulator, using action accuracy and semantic consistency metrics.","area":"Multimodal","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29247","pdf":"https://arxiv.org/pdf/2606.29247","project":null,"code":"https://github.com/VCL-HNU/SurgVLA","data":null,"hfPaper":"https://huggingface.co/papers/2606.29247"},"evidence":{"snippet":"To address this limitation, we present SurgVLA-Bench, the first comprehensive benchmark for evaluating VLA models in laparoscopic surgical robotics.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":6,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29247"},"ranking":{"90d":{"score":37,"rank":174,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates vision-language-action models in laparoscopic surgical robotics across 8 tasks (atomic, conditional, composite) on the SurRoL simulator, using action accuracy and semantic consistency metrics.","whyItMatters":"Fills the lack of standardized surgical VLA benchmarks, enabling comparison of autoregressive vs. flow-matching models and identifying physical bottlenecks like limited field of view.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b6a15db73b403a53c4833e5b37e9415d9420c71467e1d7c56a8dd26d430a164f"},"motivation":"Vision-Language-Action (VLA) models represent a promising direction for embodied intelligence in surgical robotics.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.29247","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"VCL-HNU","organizationType":"academic-lab","sourceUrl":"https://github.com/VCL-HNU/SurgVLA","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_surgwmbench_1b6538d4","familyId":"bmf_12d0da504604","name":"SurgWMBench","oneLine":"Evaluates surgical world models on short-horizon instrument motion prediction and dynamics stability from intraoperative image sequences, focusing on geometric accuracy and temporal coherence.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-08","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.08070","pdf":"https://arxiv.org/pdf/2608.08070","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.08070"},"evidence":{"snippet":"In this paper, we introduce SurgWMBench, a vision-based benchmark for short-horizon surgical motion planning and dynamics prediction.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.08070"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates surgical world models on short-horizon instrument motion prediction and dynamics stability from intraoperative image sequences, focusing on geometric accuracy and temporal coherence.","whyItMatters":"Provides a standardized protocol for motion-centric evaluation in surgical world models, addressing the lack of public datasets and metrics aligned with instrument motion planning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"fc729d568ad9455b830b5d8d22eba3661b1edafd4c99cc0f656e87c2b0c7c35b"},"motivation":"Reliable surgical planning requires models that move beyond recognizing the current surgical step or imitating expert demonstrations, and instead anticipate how instrument motion reshapes subsequent operative states.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.08070","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SurgWMBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.08070","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_a497b8fb621ca34b","familyId":"catalog_family_a497b8fb621ca34b","name":"SVG-Bench","oneLine":"SVG-Bench is an internal benchmark that comprehensively evaluates SVG generation performance. It accepts text and image inputs across build-from-scratch and edit-based tasks, using a VLM to verify rendering accuracy of the generated outputs.","description":"SVG-Bench is an internal benchmark that comprehensively evaluates SVG generation performance. It accepts text and image inputs across build-from-scratch and edit-based tasks, using a VLM to verify rendering accuracy of the generated outputs.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Multimodal","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/MiniMaxAI/MiniMax-M3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a497b8fb621ca34b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/svgbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/svg-bench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"svgBench","url":"https://benchlm.ai/benchmarks/svgbench","paperUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","year":"2026","fullName":"SVG-Bench","format":"Task success rate","tasks":"SVG generation and editing tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"svg-bench","url":"https://llm-stats.com/benchmarks/svg-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","multimodal","agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_svgeval_0c4e294c","familyId":"bmf_a28c66cbe961","name":"SVGEval","oneLine":"SVGEval is a vision-grounded benchmark for human-aligned SVG quality assessment, with multi-round labeled annotations, evaluating models on semantic, aesthetic, geometry, and layout judgments.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.01977","pdf":"https://arxiv.org/pdf/2608.01977","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.01977"},"evidence":{"snippet":"We introduce SVGEval, a vision-grounded multimodal benchmark for human-aligned SVG quality assessment.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.01977"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SVGEval is a vision-grounded benchmark for human-aligned SVG quality assessment, with multi-round labeled annotations, evaluating models on semantic, aesthetic, geometry, and layout judgments.","whyItMatters":"Addresses the lack of human-aligned evaluation for text-to-SVG generation, providing a reliable testbed for model comparison and improvement.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9f6dd5fb28a868e8d36d06bae162945d737e30f67a2293afb10dd65fd75f854e"},"motivation":"Multimodal large models are increasingly used to generate scalable vector graphics (SVG), but reliable evaluation remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026","evidence":"Accepted by ECCV 2026","evidenceUrl":"https://arxiv.org/abs/2608.01977","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026","reviewStatus":"accepted","decisionRaw":"Accepted by ECCV 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.01977","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by ECCV 2026","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_svhalluc_d54e2d9e","familyId":"bmf_d73ceabecf48","name":"SVHalluc","oneLine":"SVHalluc evaluates speech-vision hallucination in audio-visual large language models, focusing on semantic and temporal alignment between speech content and visual signals.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["eess.AS"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02642","pdf":"https://arxiv.org/pdf/2606.02642","project":"https://chenshuang-zhang.github.io/projects/svhalluc/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.02642"},"evidence":{"snippet":"To systematically study this, we introduce SVHalluc, the first comprehensive benchmark for evaluating speech-vision hallucination in audio-visual LLMs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02642"},"ranking":{},"description":"SVHalluc evaluates speech-vision hallucination in audio-visual large language models, focusing on semantic and temporal alignment between speech content and visual signals.","whyItMatters":"Addresses a critical gap in evaluating audio-visual LLMs, as prior benchmarks ignored speech-induced hallucinations. Provides a systematic method to assess model grounding, aiding in model development and selection for multimodal applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f1bed85e3bfe08a642a1a3922c8487c37af6d1c0eacd88b9a6a4d7002c669bed"},"motivation":"Despite the success of audio-visual large-language models (LLMs), they can produce plausible but ungrounded outputs, termed hallucination.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"CVPR 2026","evidence":"Accepted at CVPR 2026","evidenceUrl":"https://arxiv.org/abs/2606.02642","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"CVPR 2026","reviewStatus":"accepted","decisionRaw":"Accepted at CVPR 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.02642","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at CVPR 2026","level":"author-claim"}]}],"publishers":[{"name":"Chenshuang Zhang et al.","organizationType":"academic-lab","sourceUrl":"https://chenshuang-zhang.github.io/projects/svhalluc/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_svi-bench_918464db","familyId":"bmf_17a1b917580e","name":"SVI-Bench","oneLine":"SVI-Bench evaluates vision-language models on strategic video intelligence using sports as a microworld. It includes 9 tasks across 4 pillars: perception, reasoning, simulation, and agency, using basketball, soccer, and hockey videos with annotated actions and reports.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.31529","pdf":"https://arxiv.org/pdf/2605.31529","project":null,"code":"https://github.com/Texaser/SVI-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2605.31529"},"evidence":{"snippet":"To bridge this gap, we introduce SVI-Bench, a large-scale benchmark that leverages team sports as a dynamic microworld, combining the complexity of real-world multi-agent interaction (10-22 agents making coordinated decisions under adversarial pressure) with the verifiability of explicit rules and definitive outcomes.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":7,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.31529"},"ranking":{},"description":"SVI-Bench evaluates vision-language models on strategic video intelligence using sports as a microworld. It includes 9 tasks across 4 pillars: perception, reasoning, simulation, and agency, using basketball, soccer, and hockey videos with annotated actions and reports.","whyItMatters":"Existing video benchmarks lack verifiable ground truth for causal and strategic reasoning. SVI-Bench combines real-world multi-agent complexity with verifiable rules and outcomes, enabling evaluation of higher-level cognitive capabilities in video understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"97e6cb3853ee8765a4d34720db6c57d4d38cdd6b9cd140e15205201be13bfa7b"},"motivation":"True video intelligence demands more than recognizing what is visible: it requires reasoning about why events unfold, predicting what would change under different conditions, and deciding what to do next.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.31529","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MVP Group","organizationType":"academic-lab","sourceUrl":"https://github.com/Texaser/SVI-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_009dc944c87b90dd","familyId":"catalog_family_009dc944c87b90dd","name":"SWE Atlas - Codebase QnA","oneLine":"SWE Atlas - Codebase QnA evaluates a model's ability to answer questions about real codebases, measuring repository-level comprehension and the ability to reason about code structure, behavior, and intent across an entire project.","description":"SWE Atlas - Codebase QnA evaluates a model's ability to answer questions about real codebases, measuring repository-level comprehension and the ability to reason about code structure, behavior, and intent across an entire project.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-atlas-codebase-qna","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_009dc944c87b90dd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-atlas-codebase-qna"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-atlas-codebase-qna","url":"https://llm-stats.com/benchmarks/swe-atlas-codebase-qna","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_fb29b433a8cdf099","familyId":"catalog_family_fb29b433a8cdf099","name":"SWE Atlas - Test Writing","oneLine":"SWE Atlas - Test Writing evaluates a model's ability to author meaningful tests for real-world software projects, measuring how well agents can understand code and produce correct, useful test coverage.","description":"SWE Atlas - Test Writing evaluates a model's ability to author meaningful tests for real-world software projects, measuring how well agents can understand code and produce correct, useful test coverage.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-atlas-test-writing","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fb29b433a8cdf099"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-atlas-test-writing"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-atlas-test-writing","url":"https://llm-stats.com/benchmarks/swe-atlas-test-writing","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_c623626e37921098","familyId":"catalog_family_c623626e37921098","name":"SWE Multilingual","oneLine":"A multilingual software-engineering benchmark for real-world code issue resolution across multiple programming languages.","description":"A multilingual software-engineering benchmark for real-world code issue resolution across multiple programming languages.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Multilingual"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.minimax.io/news/minimax-m27-en","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c623626e37921098"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/swe-bench-multilingual"},{"role":"benchlm","url":"https://benchlm.ai/benchmarks/swe-bench-multilingual"}],"catalogSources":[{"catalog":"benchlm","sourceId":"sweMultilingual","url":"https://benchlm.ai/benchmarks/swe-bench-multilingual","paperUrl":"https://www.minimax.io/news/minimax-m27-en","year":"2026","fullName":"SWE Multilingual","format":"Repository task completion","tasks":"Multilingual software-engineering tasks","successorKey":null},{"catalog":"benchlm","sourceId":"sweMultilingual","url":"https://benchlm.ai/benchmarks/swe-bench-multilingual","paperUrl":"https://www.swebench.com/multilingual","year":"2025","fullName":"SWE-bench Multilingual","format":"Multi-language code patch generation","tasks":"300 problems across 9 languages","successorKey":null}],"catalogCategories":["coding","multilingual"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_d773fe5a7b8e870a","familyId":"catalog_family_d773fe5a7b8e870a","name":"SWE Multimodal","oneLine":"A multimodal variant of SWE-bench that adds visual context such as screenshots and design mockups to software engineering issue descriptions.","description":"A multimodal variant of SWE-bench that adds visual context such as screenshots and design mockups to software engineering issue descriptions.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.swebench.com/multimodal","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d773fe5a7b8e870a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/swe-bench-multimodal"}],"catalogSources":[{"catalog":"benchlm","sourceId":"sweMultimodal","url":"https://benchlm.ai/benchmarks/swe-bench-multimodal","paperUrl":"https://www.swebench.com/multimodal","year":"2025","fullName":"SWE-bench Multimodal","format":"Code patch generation with visual context","tasks":"Multimodal software engineering tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_swe-refactor-bench_45bb76b0","familyId":"bmf_e82c050bd0ea","name":"SWE Refactor Bench","oneLine":"SWE Refactor Bench is a benchmark for evaluating coding agents on whole-repository stack migrations. It comprises 20 migrations covering 4 types of technical debt, with a three-stage evaluation protocol measuring migration completeness and behavioral correctness: Migration Audit, Behavioral Tests, and Agentic Verification.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.23564v1","pdf":"https://arxiv.org/pdf/2608.23564v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To address this problem, we introduce SWE Refactor Bench, a benchmark comprising 20 whole-repository migrations, covering 4 kinds of technical debt.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23564"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SWE Refactor Bench is a benchmark for evaluating coding agents on whole-repository stack migrations. It comprises 20 migrations covering 4 types of technical debt, with a three-stage evaluation protocol measuring migration completeness and behavioral correctness: Migration Audit, Behavioral Tests, and Agentic Verification.","whyItMatters":"The benchmark addresses the gap in evaluating code migration beyond mere test passing, preventing shortcut solutions. It provides a rigorous testbed for developing reliable coding agents for long-horizon refactoring tasks, with findings on agent capability across migration categories.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"724748fbcbf6cfeb487c90ef3343c92676cf83df84f0720b6b9750afca9fe00f"},"motivation":"Modern software systems accumulate technical debt over decades of development, which makes migration expensive and largely manual.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with a rigorous multi-stage protocol and results from frontier models.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce SWE Refactor Bench, a benchmark comprising 20 whole-repository migrations"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.23564v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":60,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses the trending area of AI coding agents with a robust evaluation, likely attracting significant interest from developers and researchers."},"evaluationMode":"score_submission","publishers":[{"name":"SWE Refactor Bench Team","organizationType":"academic-lab","sourceUrl":"http://arxiv.org/abs/2608.23564v1","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"sweRefactorBench","url":"https://benchlm.ai/benchmarks/swe-refactor-bench","paperUrl":"https://arxiv.org/abs/2608.23564","year":"2026","fullName":"SWE Refactor Bench: Can Coding Agents Complete a Long-Horizon, Whole-Repository Stack Migration?","format":"Migration audit, frozen behavioral checks, and agentic verification","tasks":"20 whole-repository stack migrations","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0},{"id":"catalog_223ba126f40f8d4b","familyId":"catalog_family_223ba126f40f8d4b","name":"SWE-Atlas","oneLine":"SWE-Atlas is a software engineering benchmark focused on debugging, evaluating a model's ability to localize and fix bugs in real-world codebases.","description":"SWE-Atlas is a software engineering benchmark focused on debugging, evaluating a model's ability to localize and fix bugs in real-world codebases.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-atlas","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_223ba126f40f8d4b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-atlas"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-atlas","url":"https://llm-stats.com/benchmarks/swe-atlas","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_3d4d439dc6d1860c","familyId":"catalog_family_3d4d439dc6d1860c","name":"SWE-Atlas Refactoring","oneLine":"A Scale SWE-Atlas software-engineering agent benchmark focused on refactoring tasks.","description":"A Scale SWE-Atlas software-engineering agent benchmark focused on refactoring tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://labs.scale.com/papers/sweatlas","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3d4d439dc6d1860c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/sweatlasrefactoring"}],"catalogSources":[{"catalog":"benchlm","sourceId":"sweAtlasRefactoring","url":"https://benchlm.ai/benchmarks/sweatlasrefactoring","paperUrl":"https://labs.scale.com/papers/sweatlas","year":"2026","fullName":"SWE-Atlas Refactoring","format":"Refactoring score with confidence intervals","tasks":"SWE-Atlas refactoring tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"lib_swe_bench","familyId":"family_swe_bench","name":"SWE-bench","oneLine":"Established benchmark family · Code & Software.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code & Software"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2023-01-01","releaseDatePrecision":"year","firstRelease":{"year":2023,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2310.06770","pdf":null,"project":"https://www.swebench.com/","code":"https://github.com/SWE-bench/SWE-bench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_swe_bench"},"ranking":{},"recordType":"family","aliases":["SWEbench"],"sourceAttribution":[{"role":"official-project","url":"https://www.swebench.com/"}],"adoptionRefs":["openai-gpt5","anthropic-claude4","google-gemini25"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"lib_swe_bench_multilingual","familyId":"family_swe_bench","name":"SWE-bench Multilingual","oneLine":"Established benchmark variant · Code & Software.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code & Software"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2410.03859","pdf":null,"project":"https://www.swebench.com/multilingual.html","code":"https://github.com/SWE-bench/SWE-bench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_swe_bench_multilingual"},"ranking":{},"recordType":"variant","aliases":["SWE-bench Multilingual"],"sourceAttribution":[{"role":"benchmark-paper","url":"https://arxiv.org/abs/2410.03859"}],"adoptionRefs":["anthropic-claude4"],"modelReportReferences":[{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"variantOf":"lib_swe_bench","capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"swe-bench-multilingual","url":"https://llm-stats.com/benchmarks/swe-bench-multilingual","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","code"],"catalogModelCount":39,"catalogStarCount":0},{"id":"catalog_a425fce4110b67d7","familyId":"catalog_family_a425fce4110b67d7","name":"SWE-bench Multimodal","oneLine":"SWE-Bench Multimodal extends SWE-Bench to evaluate language models on software engineering tasks that involve visual inputs such as screenshots, UI mockups, and diagrams alongside code understanding.","description":"SWE-Bench Multimodal extends SWE-Bench to evaluate language models on software engineering tasks that involve visual inputs such as screenshots, UI mockups, and diagrams alongside code understanding.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Reasoning","Agents","Code","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.swebench.com/multimodal","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a425fce4110b67d7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/swe-bench-multimodal"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-bench-multimodal"}],"catalogSources":[{"catalog":"benchlm","sourceId":"sweMultimodal","url":"https://benchlm.ai/benchmarks/swe-bench-multimodal","paperUrl":"https://www.swebench.com/multimodal","year":"2025","fullName":"SWE-bench Multimodal","format":"Code patch generation with visual context","tasks":"Multimodal software engineering tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"swe-bench-multimodal","url":"https://llm-stats.com/benchmarks/swe-bench-multimodal","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","agents","code","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"lib_swe_bench_pro","familyId":"family_swe_bench","name":"SWE-bench Pro","oneLine":"Established benchmark variant · Code & Software.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code & Software"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2025-01-01","releaseDatePrecision":"year","firstRelease":{"year":2025,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2509.16941","pdf":null,"project":"https://scale.com/leaderboard/swe_bench_pro_public","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_swe_bench_pro"},"ranking":{},"recordType":"variant","aliases":["SWE-Bench Pro"],"sourceAttribution":[{"role":"benchmark-paper","url":"https://arxiv.org/abs/2509.16941"}],"adoptionRefs":[],"modelReportReferences":[],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"variantOf":"lib_swe_bench","capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"swePro","url":"https://benchlm.ai/benchmarks/swe-bench-pro","paperUrl":"https://arxiv.org/abs/2509.16941","year":"2025","fullName":"SWE-bench Pro","format":"Repository task completion","tasks":"1,865 repository problems","successorKey":null},{"catalog":"llm-stats","sourceId":"swe-bench-pro","url":"https://llm-stats.com/benchmarks/swe-bench-pro","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","reasoning","agents","code"],"catalogModelCount":51,"catalogStarCount":0},{"id":"bm_swe-bench_6afbfb26","familyId":"bmf_7e98773a0ce5","name":"SWE-Bench ProMax","oneLine":"A multilingual code refactoring benchmark with 170 instances drawn from real commits across seven programming languages. Evaluates AI agents on large-scale refactoring tasks averaging 11.4 modified files and 261.6 lines of code, using manually curated issue descriptions and test suites.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation"],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09802","pdf":"https://arxiv.org/pdf/2608.09802","project":null,"code":null,"data":"https://huggingface.co/datasets/swe-bench-promax/SWE-Bench-ProMax","hfPaper":"https://huggingface.co/papers/2608.09802"},"evidence":{"snippet":"We introduce SWE-Bench ProMax, an expert-curated, multilingual code refactoring benchmark of 170 instances drawn from real commits across seven programming languages (Python, Java, TypeScript, Go, C, C++, and Rust).","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":133,"hfDailySubmittedAt":"2026-08-11T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":726,"hfDatasetLikes":1},"source":{"type":"arxiv","id":"2608.09802"},"ranking":{"30d":{"score":64,"rank":7,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":7,"datasetRankPopulation":30},"90d":{"score":57,"rank":30,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":20,"datasetRankPopulation":66}},"description":"A multilingual code refactoring benchmark with 170 instances drawn from real commits across seven programming languages. Evaluates AI agents on large-scale refactoring tasks averaging 11.4 modified files and 261.6 lines of code, using manually curated issue descriptions and test suites.","whyItMatters":"Existing software engineering benchmarks face saturation and quality issues, with flawed tests and training data leakage. This benchmark provides a more challenging and realistic refactoring task set with rigorous curation, offering a robust measure of agent capability for long-horizon coding tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fa4c4a74da32976391c428ed62af680b43ba9fd7cdf6d17b40d422e01bcd079f"},"motivation":"As AI coding agents take on increasingly complex, long-horizon software engineering tasks, existing benchmarks are rapidly saturating and their evaluation quality has come under serious scrutiny: a recent audit found that nearly 60% of unsolved SWE-bench Verified instances contain flawed tests -- either overly narrow tests that reject correct solutions or overly broad tests that check unstated requirements -- and th…","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09802","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"SWE-Bench-ProMax Team","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/datasets/swe-bench-promax/SWE-Bench-ProMax","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_swe-bench_536d758e","familyId":"bmf_c0124d8dc037","name":"SWE-bench Science","oneLine":"A repository-level benchmark with 119 tasks from 98 GitHub repositories across 20 scientific domains, organized into issue-driven, expert-exploratory, and engineering-integration paradigms.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Code"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-21","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.19799","pdf":"https://arxiv.org/pdf/2608.19799","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce \\textbf{SWE-bench Science}, a repository-level benchmark for scientific software engineering comprising 119 tasks from 98 GitHub repositories across 20 scientific domains.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":62,"hfDailySubmittedAt":"2026-08-21T00:00:00.000Z","githubStars":61,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.19799"},"ranking":{"30d":{"score":89,"rank":5,"coverage":0.85,"confidence":"High"},"90d":{"score":79,"rank":24,"coverage":0.7,"confidence":"Medium"}},"description":"A repository-level benchmark with 119 tasks from 98 GitHub repositories across 20 scientific domains, organized into issue-driven, expert-exploratory, and engineering-integration paradigms.","whyItMatters":"Evaluates coding agents on scientific software failures, identifying failure modes and the nuanced role of scientific knowledge in repair tasks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"18d9f5c702f9c554acac0e9461d9466d8de63a44923c0faca5e4025fdb22d3c8"},"motivation":"Software increasingly functions as part of the scientific instrument itself, making failures in scientific code capable of compromising not only program behavior but also the evidence underlying scientific conclusions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The paper introduces a formal benchmark with a clear task structure and public repository data, enabling community reuse for evaluating coding agents in science.","canonicalNameSource":"paper_title","canonicalNameEvidence":"SWE-bench Science: Can Coding Agents Resolve Engineering Tasks in Science?"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.19799","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-21T04:30:09.569171Z"},"attentionForecast":{"score":85,"confidence":"Medium","horizon":"7d","reason":"The benchmark extends the popular SWE-bench framework to scientific domains, addressing a critical gap with strong empirical findings."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"lib_swe_bench_verified","familyId":"family_swe_bench","name":"SWE-bench Verified","oneLine":"Established benchmark variant · Code & Software.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code & Software"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://openai.com/index/introducing-swe-bench-verified/","pdf":null,"project":"https://openai.com/index/introducing-swe-bench-verified/","code":"https://github.com/SWE-bench/SWE-bench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_swe_bench_verified"},"ranking":{},"recordType":"variant","aliases":["SWE-bench Verified subset"],"sourceAttribution":[{"role":"official-release","url":"https://openai.com/index/introducing-swe-bench-verified/"}],"adoptionRefs":["openai-gpt5","anthropic-claude4","google-gemini25"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"},{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"variantOf":"lib_swe_bench","capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"sweVerified","url":"https://benchlm.ai/benchmarks/swe-bench-verified","paperUrl":"https://arxiv.org/abs/2310.06770","year":"2024","fullName":"Software Engineering Benchmark Verified","format":"Code patch generation","tasks":"500 verified issues","successorKey":null},{"catalog":"benchlm","sourceId":"sweVerifiedArcee","url":"https://benchlm.ai/benchmarks/sweverifiedarcee","paperUrl":"https://www.arcee.ai/blog/trinity-large-thinking","year":"2026","fullName":"SWE-bench Verified (mini-swe-agent-v2)","format":"Agent scaffold benchmark","tasks":"Repository task completion","successorKey":null},{"catalog":"llm-stats","sourceId":"swe-bench-verified","url":"https://llm-stats.com/benchmarks/swe-bench-verified","datasetSlug":"swe-bench-verified","versionCount":2,"subsetCount":1,"rowCount":500,"updatedAt":"2026-06-12T15:03:33.082502+00:00","community":true}],"catalogCategories":["coding","reasoning","frontend development","code"],"catalogModelCount":111,"catalogStarCount":8},{"id":"catalog_99b76d115af92ce1","familyId":"catalog_family_99b76d115af92ce1","name":"SWE-bench Verified (Agentic Coding)","oneLine":"SWE-bench Verified is a human-filtered subset of 500 software engineering problems drawn from real GitHub issues across 12 popular Python repositories. Given a codebase and an issue description, language models are tasked with generating patches that resolve the described problems. This benchmark evaluates AI's real-world agentic coding skills by requiring models to navigate complex codebases, understand software engineering problems, and coordinate changes across multiple functions, classes, and files to fix well-defined issues with clear descriptions.","description":"SWE-bench Verified is a human-filtered subset of 500 software engineering problems drawn from real GitHub issues across 12 popular Python repositories. Given a codebase and an issue description, language models are tasked with generating patches that resolve the described problems. This benchmark evaluates AI's real-world agentic coding skills by requiring models to navigate complex codebases, understand software engineering problems, and coordinate changes across multiple functions, classes, and files to fix well-defined issues with clear descriptions.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-bench-verified-(agentic-coding)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_99b76d115af92ce1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-bench-verified-(agentic-coding)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-bench-verified-(agentic-coding)","url":"https://llm-stats.com/benchmarks/swe-bench-verified-(agentic-coding)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_1896e47a987f951c","familyId":"catalog_family_1896e47a987f951c","name":"SWE-bench Verified (Agentless)","oneLine":"A human-validated subset of SWE-bench that evaluates language models' ability to resolve real-world GitHub issues using an agentless approach. The benchmark tests models on software engineering problems requiring understanding and coordinating changes across multiple functions, classes, and files simultaneously.","description":"A human-validated subset of SWE-bench that evaluates language models' ability to resolve real-world GitHub issues using an agentless approach. The benchmark tests models on software engineering problems requiring understanding and coordinating changes across multiple functions, classes, and files simultaneously.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-bench-verified-(agentless)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1896e47a987f951c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-bench-verified-(agentless)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-bench-verified-(agentless)","url":"https://llm-stats.com/benchmarks/swe-bench-verified-(agentless)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_42c12ab99c422695","familyId":"catalog_family_42c12ab99c422695","name":"SWE-bench Verified (Multiple Attempts)","oneLine":"SWE-bench Verified is a human-validated subset of 500 test samples from the original SWE-bench dataset that evaluates AI systems' ability to automatically resolve real GitHub issues in Python repositories. Given a codebase and issue description, models must edit the code to successfully resolve the problem, requiring understanding and coordination of changes across multiple functions, classes, and files. The Verified version provides more reliable evaluation through manual validation of test samples.","description":"SWE-bench Verified is a human-validated subset of 500 test samples from the original SWE-bench dataset that evaluates AI systems' ability to automatically resolve real GitHub issues in Python repositories. Given a codebase and issue description, models must edit the code to successfully resolve the problem, requiring understanding and coordination of changes across multiple functions, classes, and files. The Verified version provides more reliable evaluation through manual validation of test samples.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-bench-verified-(multiple-attempts)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_42c12ab99c422695"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-bench-verified-(multiple-attempts)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-bench-verified-(multiple-attempts)","url":"https://llm-stats.com/benchmarks/swe-bench-verified-(multiple-attempts)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_swe-explore_ece5d09b","familyId":"bmf_1b67c9f91522","name":"SWE-Explore","oneLine":"SWE-Explore evaluates repository exploration by coding agents: given a repository and an issue, an explorer returns a ranked list of relevant code regions under a fixed line budget. Ground truth is line-level, derived from successful repair trajectories. Coverage, ranking, and context-efficiency metrics are scored.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07297","pdf":"https://arxiv.org/pdf/2606.07297","project":null,"code":"https://github.com/Qiushao-E/SWE-Explore-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.07297"},"evidence":{"snippet":"In this paper, we introduce SWE-Explore, a benchmark that isolates the evaluation of repository exploration, a critical capability of coding agents.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":123,"hfDailySubmittedAt":"2026-06-09T00:00:00.000Z","githubStars":42,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07297"},"ranking":{"90d":{"score":56,"rank":32,"coverage":0.7,"confidence":"Medium"}},"description":"SWE-Explore evaluates repository exploration by coding agents: given a repository and an issue, an explorer returns a ranked list of relevant code regions under a fixed line budget. Ground truth is line-level, derived from successful repair trajectories. Coverage, ranking, and context-efficiency metrics are scored.","whyItMatters":"Existing repository-level benchmarks treat coding tasks as a single resolved/unresolved outcome, obscuring whether an agent locates the right context. SWE-Explore isolates exploration quality, enabling targeted evaluation of retrieval and localization capabilities that precede patch generation. Its metrics track downstream repair behavior, offering practical value for improving agent design.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1648e19fc75788d56e640247218b1b7764bbe77aae0bea5d3e5143dfb337fcc2"},"motivation":"Repository-level coding benchmarks such as SWE-bench have driven a rapid surge in the capabilities of coding agents.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07297","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_10a5cb2d197c84a9","familyId":"catalog_family_10a5cb2d197c84a9","name":"SWE-fficiency","oneLine":"SWE-fficiency is an open-source benchmark and workflow that evaluates language models on optimizing the runtime efficiency of real-world software engineering tasks, measuring how well agents can improve code performance autonomously.","description":"SWE-fficiency is an open-source benchmark and workflow that evaluates language models on optimizing the runtime efficiency of real-world software engineering tasks, measuring how well agents can improve code performance autonomously.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-fficiency","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_10a5cb2d197c84a9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-fficiency"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-fficiency","url":"https://llm-stats.com/benchmarks/swe-fficiency","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_swe-interact_046bbdbf","familyId":"bmf_93acd7a96f92","name":"SWE-INTERACT","oneLine":"SWE-Interact evaluates coding agents on multi-turn, interactive software engineering tasks where a simulated user provides vague instructions, reveals requirements progressively, and gives feedback. The benchmark comprises 75 tasks and measures agents' ability to discover user intent, adapt to evolving requirements, and build on prior work.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Code"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.30573","pdf":"https://arxiv.org/pdf/2606.30573","project":null,"code":"https://github.com/scaleapi/SWE-Interact","data":null,"hfPaper":"https://huggingface.co/papers/2606.30573"},"evidence":{"snippet":"We introduce SWE-Interact, a new testbed for evaluating coding agents on multi-turn, interactive, user-driven software engineering tasks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":"2026-07-01T00:00:00.000Z","githubStars":25,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.30573"},"ranking":{"90d":{"score":45,"rank":109,"coverage":0.7,"confidence":"Medium"}},"description":"SWE-Interact evaluates coding agents on multi-turn, interactive software engineering tasks where a simulated user provides vague instructions, reveals requirements progressively, and gives feedback. The benchmark comprises 75 tasks and measures agents' ability to discover user intent, adapt to evolving requirements, and build on prior work.","whyItMatters":"Existing SWE benchmarks focus on single-turn autonomous implementation, but real developer workflows are interactive. SWE-Interact fills the gap by measuring performance on long-horizon, user-driven tasks, showing that strong single-turn performance does not reliably transfer. This provides a more realistic evaluation axis for coding agents and guides development of models that can collaborate effectively with users.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f9fc79903ecb54517711cf6a755aafb2701ab8e2f8bdce223851471f86fae8cb"},"motivation":"We introduce SWE-Interact, a new testbed for evaluating coding agents on multi-turn, interactive, user-driven software engineering tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.30573","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Scale AI","organizationType":"company-research-lab","sourceUrl":"https://github.com/scaleapi/SWE-Interact","role":"benchmark-publisher"}],"capabilityGroups":["Agents","Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_bbef19a7e22e351d","familyId":"catalog_family_bbef19a7e22e351d","name":"SWE-Lancer","oneLine":"A benchmark for evaluating large language models on real-world freelance software engineering tasks from Upwork. Contains over 1,400 tasks valued at $1 million USD total, ranging from $50 bug fixes to $32,000 feature implementations. Includes both independent engineering tasks graded via end-to-end tests and managerial tasks assessed against original engineering managers' choices.","description":"A benchmark for evaluating large language models on real-world freelance software engineering tasks from Upwork. Contains over 1,400 tasks valued at $1 million USD total, ranging from $50 bug fixes to $32,000 feature implementations. Includes both independent engineering tasks graded via end-to-end tests and managerial tasks assessed against original engineering managers' choices.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-lancer","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bbef19a7e22e351d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-lancer"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-lancer","url":"https://llm-stats.com/benchmarks/swe-lancer","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","code"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_a8c2a6f6c2e62be3","familyId":"catalog_family_a8c2a6f6c2e62be3","name":"SWE-Lancer (IC-Diamond subset)","oneLine":"SWE-Lancer (IC-Diamond subset) is a benchmark of real-world freelance software engineering tasks from Upwork, ranging from $50 bug fixes to $32,000 feature implementations. It evaluates AI models on independent engineering tasks using end-to-end tests triple-verified by experienced software engineers, and includes managerial tasks where models choose between technical implementation proposals.","description":"SWE-Lancer (IC-Diamond subset) is a benchmark of real-world freelance software engineering tasks from Upwork, ranging from $50 bug fixes to $32,000 feature implementations. It evaluates AI models on independent engineering tasks using end-to-end tests triple-verified by experienced software engineers, and includes managerial tasks where models choose between technical implementation proposals.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-lancer-(ic-diamond-subset)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a8c2a6f6c2e62be3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-lancer-(ic-diamond-subset)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-lancer-(ic-diamond-subset)","url":"https://llm-stats.com/benchmarks/swe-lancer-(ic-diamond-subset)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","code"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_swe-marathon_2e8e782b","familyId":"bmf_fbbe5c4b7d4a","name":"SWE-Marathon","oneLine":"SWE-Marathon evaluates AI agents on 20 ultra-long-horizon software engineering tasks, each with a unique executable environment, a human-written reference solution, and a multi-layer verification suite. Tasks average 27.2M tokens per logged agent attempt, requiring sustained progress over hours and millions of tokens.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07682","pdf":"https://arxiv.org/pdf/2606.07682","project":"https://swe-marathon.org/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07682"},"evidence":{"snippet":"We introduce SWE-Marathon, a benchmark of 20 long-horizon tasks spanning software engineering and adjacent technical domains.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07682"},"ranking":{"90d":{"score":45,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SWE-Marathon evaluates AI agents on 20 ultra-long-horizon software engineering tasks, each with a unique executable environment, a human-written reference solution, and a multi-layer verification suite. Tasks average 27.2M tokens per logged agent attempt, requiring sustained progress over hours and millions of tokens.","whyItMatters":"Existing agent benchmarks focus on short tasks, limiting measurement of planning, long-context understanding, and memory. SWE-Marathon addresses the gap by providing a longer-horizon evaluation that exposes practical limitations in agent autonomy and highlights failure modes like reward hacking, informing development of more robust agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4a8516fea7e8c588b9b3aa7d8155f8d86deddf6e4bbc0af148466d0996221ea2"},"motivation":"AI agents are increasingly expected to complete long-horizon workflows that require sustained progress over hours, millions of tokens, and complex environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07682","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"swe-marathon.org","organizationType":"community","sourceUrl":"https://swe-marathon.org/","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"sweMarathon","url":"https://benchlm.ai/benchmarks/swemarathon","paperUrl":"https://www.swe-marathon.org/","year":"2026","fullName":"SWE-Marathon","format":"Task resolution and trajectory review","tasks":"20 multi-hour software engineering tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"swe-marathon","url":"https://llm-stats.com/benchmarks/swe-marathon","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["external","agents","code"],"catalogModelCount":4,"catalogStarCount":0},{"id":"catalog_d59473cf8cd9e648","familyId":"catalog_family_d59473cf8cd9e648","name":"SWE-MM","oneLine":"SWE-MM evaluates software-engineering agents on repository tasks that require understanding both source code and visual evidence.","description":"SWE-MM evaluates software-engineering agents on repository tasks that require understanding both source code and visual evidence.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Agents","Code","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-mm","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d59473cf8cd9e648"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-mm"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-mm","url":"https://llm-stats.com/benchmarks/swe-mm","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","agents","code","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_swe-mutation_d27a11bf","familyId":"bmf_a72ddf5211a2","name":"SWE-Mutation","oneLine":"SWE-Mutation evaluates LLM-generated test suites in software engineering by using 2,636 mutated variants derived from 800 original instances across nine programming languages, measuring verification and detection rates.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Code"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22175","pdf":"https://arxiv.org/pdf/2605.22175","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.22175"},"evidence":{"snippet":"As a first step toward constructing high-quality test suites, we introduce SWE-Mutation, a benchmark for evaluating LLM-generated test suites.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.22175"},"ranking":{},"description":"SWE-Mutation evaluates LLM-generated test suites in software engineering by using 2,636 mutated variants derived from 800 original instances across nine programming languages, measuring verification and detection rates.","whyItMatters":"High-quality test suites are critical for program repair and reinforcement learning signals. SWE-Mutation reveals inadequacies in LLM-generated tests, guiding improvements in code generation and validation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"e441cc8ae619179d87f2496a3f9d36c1468825e6fe540339831478355b235ff4"},"motivation":"Evaluating software engineering capabilities has become a core component of modern large language models (LLMs); however, the key bottleneck hindering further scaling lies not in the scarcity of high-quality solutions, but in the lack of high-quality test suites.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"publication_reported","venue":"ACL 2026 Findings","evidence":"ACL 2026 Findings","evidenceUrl":"https://arxiv.org/abs/2605.22175","source":"arxiv-journal-reference","evidenceLevel":"strong-author-metadata","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publications":[{"venueName":"ACL 2026 Findings","publicationStatus":"published","evidence":[{"sourceType":"arxiv-journal-reference","sourceUrl":"https://arxiv.org/abs/2605.22175","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"ACL 2026 Findings","level":"strong-author-metadata"}]}],"publishers":[{"name":"SWE-Mutation Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2605.22175","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_swe-nfi_7cf5877f","familyId":"bmf_b14a6af93273","name":"SWE-NFI","oneLine":"A benchmark of 188 tasks for evaluating coding agents on non-functional improvements in Python projects, with 92 executable rules combining functional correctness and rule-based evaluation.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27409","pdf":"https://arxiv.org/pdf/2607.27409","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.27409"},"evidence":{"snippet":"In this paper, we present SWE-NFI, a benchmark for evaluating coding agents on NFIs beyond functional correctness.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27409"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark of 188 tasks for evaluating coding agents on non-functional improvements in Python projects, with 92 executable rules combining functional correctness and rule-based evaluation.","whyItMatters":"Existing coding benchmarks focus on functional correctness; this benchmark addresses the gap in evaluating behavior-preserving code quality improvements, useful for assessing real-world software engineering capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5f29664e18b98d6ce93447ff56a9a4f0b4ba0d9b77625f5d60cddf5dbc5d0c25"},"motivation":"Although coding agents have achieved impressive performance on correctness-oriented benchmarks, their ability to make behavior-preserving non-functional improvements (NFIs) remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27409","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_b7ec09b15b4763be","familyId":"catalog_family_b7ec09b15b4763be","name":"SWE-Perf","oneLine":"Software Engineering Performance benchmark measuring code optimization capabilities","description":"Software Engineering Performance benchmark measuring code optimization capabilities","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-perf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b7ec09b15b4763be"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-perf"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-perf","url":"https://llm-stats.com/benchmarks/swe-perf","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_6ab36870da0850d4","familyId":"catalog_family_6ab36870da0850d4","name":"SWE-Rebench","oneLine":"A continuously updated software engineering benchmark by Nebius using fresh GitHub issues to avoid contamination. Models are evaluated 5 times per problem under a fixed ReAct scaffolding; the Resolved Rate (best pass@1) is reported.","description":"A continuously updated software engineering benchmark by Nebius using fresh GitHub issues to avoid contamination. Models are evaluated 5 times per problem under a fixed ReAct scaffolding; the Resolved Rate (best pass@1) is reported.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://swe-rebench.com","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6ab36870da0850d4"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/swe-rebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"sweRebench","url":"https://benchlm.ai/benchmarks/swe-rebench","paperUrl":"https://swe-rebench.com","year":"2026","fullName":"SWE-Rebench","format":"Code patch generation","tasks":"Fresh GitHub issues (rolling window)","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_e12892e6d3b8f160","familyId":"catalog_family_e12892e6d3b8f160","name":"SWE-Review","oneLine":"Software Engineering Review benchmark evaluating code review capabilities","description":"Software Engineering Review benchmark evaluating code review capabilities","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swe-review","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e12892e6d3b8f160"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swe-review"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swe-review","url":"https://llm-stats.com/benchmarks/swe-review","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_swe-together_260e2683","familyId":"bmf_123ecb3dd733","name":"SWE-Together","oneLine":"SWE-Together evaluates coding agents in multi-turn interactive user sessions reconstructed from real user-agent interactions. It comprises 109 repository-level tasks with a reactive LLM-based user simulator, measuring final repository correctness and the number of corrective feedback turns.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Agents"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29957","pdf":"https://arxiv.org/pdf/2606.29957","project":null,"code":"https://github.com/Togetherbench/SWE-Together","data":null,"hfPaper":"https://huggingface.co/papers/2606.29957"},"evidence":{"snippet":"We introduce SWE-Together, a multi-turn benchmark reconstructed from real user-agent coding sessions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":14,"hfDailySubmittedAt":"2026-06-30T00:00:00.000Z","githubStars":59,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29957"},"ranking":{"90d":{"score":53,"rank":47,"coverage":0.7,"confidence":"Medium"}},"description":"SWE-Together evaluates coding agents in multi-turn interactive user sessions reconstructed from real user-agent interactions. It comprises 109 repository-level tasks with a reactive LLM-based user simulator, measuring final repository correctness and the number of corrective feedback turns.","whyItMatters":"Existing coding-agent benchmarks often evaluate static, single-turn tasks, missing the interactive nature of real coding assistance. SWE-Together provides a reproducible protocol for assessing agents as collaborators, capturing both task success and user effort, offering practical value for comparing agents in realistic settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a2b28d10dee2abdbcde94ad261c709e00c47c052a270c697ee14dce0e07a2dc2"},"motivation":"Most coding-agent benchmarks are static: an agent receives a complete task description up front and is judged only by its final code.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.29957","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Togetherbench","organizationType":"community","sourceUrl":"https://github.com/Togetherbench/SWE-Together","role":"benchmark-publisher"}],"capabilityGroups":["Agents","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_swisscrop25_6e928630","familyId":"bmf_038a2a2ba3b9","name":"SwissCrop25","oneLine":"Evaluates crop mapping models on national-scale multi-year Sentinel-2 data with fine-grained crop taxonomy and non-crop classes under leave-one-year-out protocol.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":0.8,"links":{"report":"https://arxiv.org/abs/2608.09497","pdf":"https://arxiv.org/pdf/2608.09497","project":null,"code":null,"data":"https://huggingface.co/datasets/EOA-team/SwissCrop25","hfPaper":null},"evidence":{"snippet":"We therefore introduce SwissCrop25, a national-scale crop mapping benchmark dataset spanning seven growing seasons (2019-2025).","reasonCodes":["exact coined title identity tied to benchmark evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":802,"hfDatasetLikes":2},"source":{"type":"arxiv","id":"2608.09497"},"ranking":{"30d":{"score":38,"rank":54,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":6,"datasetRankPopulation":30},"90d":{"score":44,"rank":116,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":19,"datasetRankPopulation":66}},"description":"Evaluates crop mapping models on national-scale multi-year Sentinel-2 data with fine-grained crop taxonomy and non-crop classes under leave-one-year-out protocol.","whyItMatters":"Provides a realistic operational testbed that exposes interannual distribution shifts and architecture differences hidden by conventional benchmarks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"31df337f0beac963716db90c664ee4781206ecbc782e8911324933b7ee9c9ad7"},"motivation":"Operational crop mapping requires models that generalise across years, resolve fine-grained crop taxonomies, and distinguish cropland from surrounding landscapes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"The dataset and evaluation protocol are publicly released with clear licensing and benchmarks for three models, enabling reuse.","canonicalNameSource":"abstract","canonicalNameEvidence":"We therefore introduce SwissCrop25, a national-scale crop mapping benchmark dataset"},"publication":{"status":"acceptance_claimed","venue":"ECCV 2026 Workshop TerraBytes II","evidence":"Accepted at the ECCV 2026 Workshop TerraBytes II. To appear in the workshop proceedings","evidenceUrl":"https://arxiv.org/abs/2608.09497","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-19T11:05:13.395391Z"},"venueAttempts":[{"venueName":"ECCV 2026 Workshop TerraBytes II","reviewStatus":"accepted","decisionRaw":"Accepted at the ECCV 2026 Workshop TerraBytes II. To appear in the workshop proceedings","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.09497","observedAt":"2026-08-19T11:05:13.395391Z","rawValue":"Accepted at the ECCV 2026 Workshop TerraBytes II. To appear in the workshop proceedings","level":"author-claim"}]}],"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"The ECCV workshop acceptance, public dataset, and practical relevance for operational crop mapping should attract moderate attention from remote sensing and agricultural AI communities."},"evaluationMode":"public_reusable","publishers":[{"name":"Earth Observation of Agroecosystems Team, Agroscope","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/datasets/EOA-team/SwissCrop25","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_6f5e7c2ddd66ffe5","familyId":"catalog_family_6f5e7c2ddd66ffe5","name":"SWT-Bench","oneLine":"Software Test Benchmark evaluating LLM ability to write tests for software repositories","description":"Software Test Benchmark evaluating LLM ability to write tests for software repositories","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/swt-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6f5e7c2ddd66ffe5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/swt-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"swt-bench","url":"https://llm-stats.com/benchmarks/swt-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_symbalbench_84a0aab6","familyId":"bmf_26aed8f715b6","name":"SymbalBench","oneLine":"SymbalBench evaluates automated detection of systematic misalignments in MLLM-generated image captions. It comprises 420 vision-language datasets (1.7 million image-text pairs) from natural and medical domains, each annotated with known systematic misalignments.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.15216","pdf":"https://arxiv.org/pdf/2607.15216","project":null,"code":"https://github.com/Stanford-AIMI/Symbal","data":null,"hfPaper":"https://huggingface.co/papers/2607.15216"},"evidence":{"snippet":"As our second key contribution, we introduce SymbalBench, a benchmark designed to evaluate automated methods on our proposed task.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.15216"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"SymbalBench evaluates automated detection of systematic misalignments in MLLM-generated image captions. It comprises 420 vision-language datasets (1.7 million image-text pairs) from natural and medical domains, each annotated with known systematic misalignments.","whyItMatters":"MLLM-generated captions often contain recurring errors tied to visual features, which can degrade downstream tasks. SymbalBench provides a standardized testbed to assess methods for surfacing such systematic captioning failures, aiding dataset auditing and model improvement.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0718cfa0851993fa77465aa34998a9780399ae8aa865c14932469950b3c6d10a"},"motivation":"Multimodal large language models (MLLMs) often introduce errors when generating image captions, resulting in misaligned image-text pairs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.15216","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Stanford AIMI","organizationType":"academic-lab","sourceUrl":"https://github.com/Stanford-AIMI/Symbal","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_syncred-bench_70361b01","familyId":"bmf_57e951fbab79","name":"SynCred-Bench","oneLine":"SynCred-Bench evaluates detection of AI-generated visual misinformation with synthetic credibility, using 600 AI-generated images and FP450 real-image negatives.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.03348","pdf":"https://arxiv.org/pdf/2606.03348","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03348"},"evidence":{"snippet":"We introduce SYNCRED-Bench, a benchmark of 600 AI-generated misinformation images balanced across six credible-form categories and seven fine-grained circulation styles, together with FP450, a real-image negative set for measuring false positives.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03348"},"ranking":{"90d":{"score":45,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"SynCred-Bench evaluates detection of AI-generated visual misinformation with synthetic credibility, using 600 AI-generated images and FP450 real-image negatives.","whyItMatters":"No public artifacts or reuse path are provided, and the benchmark lacks a clear scoring contract for independent runs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8181ec3f33e3d9ad04691f05c6bbaf99fe99fdadf608208920f0a39e525f122e"},"motivation":"Recent generative models can now produce visual artifacts with realistic embedded text and layouts, creating a new misinformation threat: synthetic credibility.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03348","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_synthave_eb33ab8a","familyId":"bmf_73cc3ce2801f","name":"SynthAVE","oneLine":"Presents a synthetic labeling pipeline for e-commerce attribute extraction with human validation via multi-LLM arena voting.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.07469","pdf":"https://arxiv.org/pdf/2607.07469","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.07469"},"evidence":{"snippet":"We present SynthAVE, a large-scale human-validated benchmark for attribute value extraction spanning 12,726 products across 229 product types, 792 attributes, and 4 languages (Spanish, French, Italian, German).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.07469"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Presents a synthetic labeling pipeline for e-commerce attribute extraction with human validation via multi-LLM arena voting.","whyItMatters":"Demonstrates a cost-effective method for large-scale label generation with quality control, but is not a benchmark for model comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"59716af9ba809dfe7a58a2b2996cf7370992c87378d8b6d33736185e6cc821ad"},"motivation":"Fine-tuning large language models (LLMs) for e-commerce attribute extraction requires labeled data representative across thousands of product types, attributes, and multiple languages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.07469","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_synthdocbench_7af1a05d","familyId":"bmf_3a3c4b0877bd","name":"SynthDocBench","oneLine":"SynthDocBench evaluates vision-language models on 1,788 questions over 200 synthetic long-context documents (avg. 51.1 pages) with 3,340 charts. It varies document length, layout archetype, modality composition, and question type as independent controlled factors, across chart, cross-modal, and complex subsets with deterministic ground truth.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-07-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.10400","pdf":"https://arxiv.org/pdf/2607.10400","project":null,"code":"https://github.com/ServiceNow/SynthDocBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.10400"},"evidence":{"snippet":"We introduce SynthDocBench, a fully synthetic benchmark for long-context visual document understanding that systematically controls factors including document length, layout structure, modality composition, and question type.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":71,"hfDailySubmittedAt":null,"githubStars":9,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.10400"},"ranking":{"90d":{"score":44,"rank":122,"coverage":0.7,"confidence":"Medium"}},"description":"SynthDocBench evaluates vision-language models on 1,788 questions over 200 synthetic long-context documents (avg. 51.1 pages) with 3,340 charts. It varies document length, layout archetype, modality composition, and question type as independent controlled factors, across chart, cross-modal, and complex subsets with deterministic ground truth.","whyItMatters":"Existing document benchmarks confound length, layout, and modality, obscuring specific model failure causes. SynthDocBench provides controlled attribution of failures, revealing sharp degradation with document length, positional sensitivity in the middle third, and breakdown of chart comprehension in long documents, which is valuable for targeted model improvement.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a77aa761d725adc596250e12d35b6b03c1d0b02ed127e66a50dd9ccf9c1383c2"},"motivation":"Vision language models (VLMs) have achieved strong performance on visual document understanding benchmarks such as DocVQA, ChartQA, and MMLongBench-Doc.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"COLM 2026","evidence":"29 Pages, 27 Tables, 13 Figures, Accepted at COLM 2026","evidenceUrl":"https://arxiv.org/abs/2607.10400","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"COLM 2026","reviewStatus":"accepted","decisionRaw":"29 Pages, 27 Tables, 13 Figures, Accepted at COLM 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.10400","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"29 Pages, 27 Tables, 13 Figures, Accepted at COLM 2026","level":"author-claim"}]}],"publishers":[{"name":"ServiceNow AI","organizationType":"company-research-lab","sourceUrl":"https://github.com/ServiceNow/SynthDocBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception","Long Context & Memory"],"domainScope":"general"},{"id":"bm_t-impact_38620781","familyId":"bmf_bc176ef6047b","name":"T-IMPACT","oneLine":"T-IMPACT evaluates models on severity-aware detection of manipulated news-style image-text pairs, with 98,786 examples covering pristine, image-only, text-only, and joint manipulations, alongside calibrated continuous severity scores, coarse labels, and grounding metadata.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.22339","pdf":"https://arxiv.org/pdf/2606.22339","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.22339"},"evidence":{"snippet":"We introduce T-IMPACT, a first-release severity-aware benchmark for manipulated news-style image-text pairs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.22339"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"T-IMPACT evaluates models on severity-aware detection of manipulated news-style image-text pairs, with 98,786 examples covering pristine, image-only, text-only, and joint manipulations, alongside calibrated continuous severity scores, coarse labels, and grounding metadata.","whyItMatters":"Existing multimodal manipulation benchmarks focus on authenticity or manipulation type, lacking graded impact severity. T-IMPACT fills this gap with a calibrated continuous severity signal, enabling evaluation of models' ability to judge contextual impact, which is essential for mitigating persuasive misinformation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f1424a71a9c6e791436ef4d9dda5c9105316a0b14e0877cf67e9ba1d6e9d9f8e"},"motivation":"Recent advances in vision-language models and generative editing systems have made it increasingly easy to produce persuasive multimodal misinformation by altering images, text, or both jointly.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.22339","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_t1-bench_01df3d35","familyId":"bmf_dbcfe3a64553","name":"T1-Bench","oneLine":"T1-Bench evaluates agentic systems through realistic customer-facing, multi-domain tasks. Scenarios involve multi-turn user-assistant interactions across 25 domains with varying difficulty, measuring structured reasoning, tool utilization, and conversational quality. Automatic evaluation is complemented by human judgments.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11070","pdf":"https://arxiv.org/pdf/2606.11070","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.11070"},"evidence":{"snippet":"To address these limitations, we introduce T1-Bench, a high-fidelity, comprehensive benchmark for evaluating agentic systems in realistic customer-facing, multi-domain environments, featuring interleaved scenarios that require structured reasoning across multi-turn user-assistant interactions and substantially increasing both compositional complexity and evaluative rigor across 25 domains of varying difficulty.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.11070"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"T1-Bench evaluates agentic systems through realistic customer-facing, multi-domain tasks. Scenarios involve multi-turn user-assistant interactions across 25 domains with varying difficulty, measuring structured reasoning, tool utilization, and conversational quality. Automatic evaluation is complemented by human judgments.","whyItMatters":"Existing agent benchmarks lack realism and domain diversity, limiting assessment of sustained multi-step reasoning. T1-Bench addresses this gap by providing interleaved, complex scenarios across multiple domains, enabling more accurate evaluation of agents' practical capabilities in real-world customer service settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6398f5c4a2cce8c21857c40b86c8bd3ca282bc018316f5e2ee279296e0b20e33"},"motivation":"Recent advances in reasoning and tool-calling capabilities of large language models (LLMs) have enabled increasingly capable agentic systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11070","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_c3436709ce24412e","familyId":"catalog_family_c3436709ce24412e","name":"t2-bench","oneLine":"t2-bench is a benchmark for evaluating agentic tool use capabilities, measuring how well models can select, sequence, and utilize tools to solve complex tasks. It tests autonomous planning and execution in multi-step scenarios.","description":"t2-bench is a benchmark for evaluating agentic tool use capabilities, measuring how well models can select, sequence, and utilize tools to solve complex tasks. It tests autonomous planning and execution in multi-step scenarios.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/t2-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c3436709ce24412e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/t2-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"t2-bench","url":"https://llm-stats.com/benchmarks/t2-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","tool calling"],"catalogModelCount":23,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_t2d-bench_6b92ee7c","familyId":"bmf_ece4accf1aab","name":"T2D-Bench","oneLine":"T2D-Bench evaluates LLM outputs for type 2 diabetes against evidence constraints using a multi-layer clinical-lifestyle knowledge graph, covering diagnosis, medication safety, and lifestyle conflicts across 100 structured vignettes.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24145","pdf":"https://arxiv.org/pdf/2606.24145","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.24145"},"evidence":{"snippet":"We present T2D-Bench, a reproducible benchmark and evidence-gated evaluation framework for testing whether LLM outputs satisfy explicit, graph-checkable evidence requirements.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24145"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"T2D-Bench evaluates LLM outputs for type 2 diabetes against evidence constraints using a multi-layer clinical-lifestyle knowledge graph, covering diagnosis, medication safety, and lifestyle conflicts across 100 structured vignettes.","whyItMatters":"Addresses the gap in evaluating whether LLM recommendations satisfy explicit clinical guidelines and justify lifestyle-related glycemic claims, providing a mechanism to detect unsupported omissions and improve verifier-level compliance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"15c2d1bbec2fdc0844a712ad7ee5c89bf538628056cfea5a1ff4d76fdc364dbd"},"motivation":"Large language models (LLMs) can produce clinically fluent recommendations for type 2 diabetes while failing to satisfy guideline constraints or explicitly justify lifestyle-related glycemic claims.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"as a poster at AMIA 2026 Annual Symposium","evidence":"7 pages, 2 figures, 2 tables. Accepted as a poster at AMIA 2026 Annual Symposium","evidenceUrl":"https://arxiv.org/abs/2606.24145","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"as a poster at AMIA 2026 Annual Symposium","reviewStatus":"accepted","decisionRaw":"7 pages, 2 figures, 2 tables. Accepted as a poster at AMIA 2026 Annual Symposium","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.24145","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"7 pages, 2 figures, 2 tables. Accepted as a poster at AMIA 2026 Annual Symposium","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_t2j-bench_85fea4b8","familyId":"bmf_92b4500c07b1","name":"T2J-Bench","oneLine":"T2J-Bench benchmarks codebase conversion by transferring PyTorch code to JAX under a fixed equivalence contract with three ordered verification stages.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2605.29054","pdf":"https://arxiv.org/pdf/2605.29054","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.29054"},"evidence":{"snippet":"We introduce T2J-Bench, a benchmark for codebase conversion that reformulates conversion as transfer under a fixed equivalence contract.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.29054"},"ranking":{},"description":"T2J-Bench benchmarks codebase conversion by transferring PyTorch code to JAX under a fixed equivalence contract with three ordered verification stages.","whyItMatters":"Reveals that agents overestimate success on codebase conversion, but the benchmark's data and verification harness are not publicly released.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ab69166eb290d1e70f471e6c40ad2ace2ebbfc098fc1b206de14ef93048de688"},"motivation":"Coding agents increasingly act as codebase-scale collaborators that can assist with codebase conversion, but this progress has exposed a critical weakness: agents often over-trust their own local validation routines and declare success on artifacts that satisfy surface checks while violating the semantic contracts users actually care about.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29054","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_tabquerybench_d985f8e2","familyId":"bmf_bdd17825921f","name":"TabQueryBench","oneLine":"TabQueryBench evaluates synthetic tabular data generators using SQL-shaped analytical queries as structural assessors, providing 44 reusable query templates across 49 datasets.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.DB"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-04","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.03926","pdf":"https://arxiv.org/pdf/2607.03926","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.03926"},"evidence":{"snippet":"We present TabQueryBench, a query-centric benchmark that uses SQL-shaped analytical queries as structural assessors for synthetic data fidelity.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.03926"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TabQueryBench evaluates synthetic tabular data generators using SQL-shaped analytical queries as structural assessors, providing 44 reusable query templates across 49 datasets.","whyItMatters":"Existing synthetic data evaluations focus on statistical similarity and downstream ML utility, but rarely test analytical query structure. This benchmark addresses that gap, offering a way to assess query-centric fidelity for practical data analysis use cases.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"980fc5b0a5c4dc58654070e0131a5581188e0bc525f962e85db4c94197ef6778"},"motivation":"Synthetic tabular data support use cases like data sharing, model development under access restrictions, and rapid prototyping of analytical workflows.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.03926","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_tabverse_d67e32ec","familyId":"bmf_4624dae44f51","name":"TABVERSE","oneLine":"TABVERSE is a controlled multimodal benchmark that aligns the same table content across HTML, Markdown, LaTeX, and rendered images, with question category and difficulty tags. It evaluates LLMs and VLMs on question answering, structural understanding, and structure reconstruction.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09578","pdf":"https://arxiv.org/pdf/2606.09578","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.09578"},"evidence":{"snippet":"We introduce TABVERSE, a controlled multimodal table benchmark that aligns the same table content across multiple structural formats and rendered images, with question category and difficulty tags.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09578"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"TABVERSE is a controlled multimodal benchmark that aligns the same table content across HTML, Markdown, LaTeX, and rendered images, with question category and difficulty tags. It evaluates LLMs and VLMs on question answering, structural understanding, and structure reconstruction.","whyItMatters":"Existing table benchmarks conflate content, format, and modality, obscuring the impact of representation choice. TABVERSE enables isolation of representation effects, providing practical guidance for selecting robust table formats for downstream applications and evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"44564f038262799f85352a80ee39a3c6ef3727ff6cada747f12f71fedd6e7ea5"},"motivation":"Large Language Models (LLMs) and Vision-Language Models (VLMs) are increasingly evaluated on table reasoning tasks, but the role of table representation remains under-explored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09578","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_tactidex_18e67e16","familyId":"bmf_e1d0d3e790ce","name":"TactiDex","oneLine":"TactiDex is a real-world tactile-guided benchmark for dexterous manipulation, aligning whole-hand tactile signals with kinematic and object states. It provides standardized evaluation metrics and a framework for tactile-driven transfer, including experiments on single and bimanual tasks.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.09190","pdf":"https://arxiv.org/pdf/2607.09190","project":"https://tactidex.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.09190"},"evidence":{"snippet":"To address this, we introduce TactiDex, a real-world tactile-guided benchmark specifically designed to move dexterous manipulation beyond kinematic mimicry toward contact-level human-likeness.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.09190"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TactiDex is a real-world tactile-guided benchmark for dexterous manipulation, aligning whole-hand tactile signals with kinematic and object states. It provides standardized evaluation metrics and a framework for tactile-driven transfer, including experiments on single and bimanual tasks.","whyItMatters":"Tactile feedback is essential for human-like dexterous manipulation, yet existing benchmarks focus on kinematic imitation. TactiDex addresses this gap by providing a benchmark that emphasizes contact-level human-likeness, enabling physically grounded robot execution.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0c6ae15678ee84d9a6112c5c5634af9feeb5fc7e7f58105b01022e2dfa54a959"},"motivation":"Tactile feedback is fundamental to Hand-Object Interaction (HOI), governing contact formation, force regulation, and stable manipulation, making it essential for achieving true human-like dexterous manipulation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.09190","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TactiDex Team","organizationType":"academic-lab","sourceUrl":"https://tactidex.github.io/","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_tacverse_8b86d6a1","familyId":"bmf_876804079426","name":"TacVerse","oneLine":"TacVerse is a multi-sensor dataset and benchmark for cross-sensor vision-based tactile perception, containing 106,800 tactile images from seven vision-based tactile sensors. It supports shape classification, grating classification, and force regression tasks, with evaluation under within-sensor, zero-shot cross-sensor, and few-shot adaptation settings.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.25877","pdf":"https://arxiv.org/pdf/2606.25877","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.25877"},"evidence":{"snippet":"We present TacVerse, a multi-sensor dataset and benchmark for cross-sensor vision-based tactile perception.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.25877"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TacVerse is a multi-sensor dataset and benchmark for cross-sensor vision-based tactile perception, containing 106,800 tactile images from seven vision-based tactile sensors. It supports shape classification, grating classification, and force regression tasks, with evaluation under within-sensor, zero-shot cross-sensor, and few-shot adaptation settings.","whyItMatters":"TacVerse fills a gap in evaluating generalization across tactile sensor designs, which is crucial for real-world robot deployment. It enables systematic study of sensor shift, data-efficient adaptation, and self-supervised learning in tactile perception, providing a controlled testbed for improving cross-sensor robustness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"56e813bce40f06b11b51293f939370ed7a5a66828574ff70403ea2c4ba00709e"},"motivation":"Vision-based tactile sensors (VBTSs) enable robots to infer contact geometry and force-related cues by imaging deformation through an internal camera, yet generalisation across sensor designs remains poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.25877","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_tada-bench_769ea7c2","familyId":"bmf_0784ab615c0b","name":"TadA-Bench","oneLine":"TadA-Bench is a fixed-data replay benchmark derived from 31 wet-lab rounds of TadA directed evolution. Models rank ~1M protein, DNA, or RNA sequence variants appearing only in later rounds, with scores as Spearman, Recall@10%, and nDCG@10%.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["q-bio.QM"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02624","pdf":"https://arxiv.org/pdf/2606.02624","project":null,"code":"https://github.com/shiyegao/TadABench-1M","data":"https://huggingface.co/datasets/JinGao/TadABench-1M","hfPaper":"https://huggingface.co/papers/2606.02624"},"evidence":{"snippet":"We introduce TadA-Bench, a million-variant wet-lab replay benchmark from 31 TadA directed-evolution rounds for future-round discovery toward agentic protein engineering.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":4,"githubScope":"benchmark_repo","hfDatasetDownloads":628,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2606.02624"},"ranking":{},"description":"TadA-Bench is a fixed-data replay benchmark derived from 31 wet-lab rounds of TadA directed evolution. Models rank ~1M protein, DNA, or RNA sequence variants appearing only in later rounds, with scores as Spearman, Recall@10%, and nDCG@10%.","whyItMatters":"Standard random-split evaluation overestimates performance for iterative candidate prioritization. TadA-Bench provides a chronological replay protocol to measure whether models can transfer from earlier experimental rounds to future ones, a core requirement for agentic protein engineering.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a9967bd21cdc02003589cb6cb7db5ed19dcd7a95ea2a111f18834ae8f98ec307"},"motivation":"AI for scientific discovery is entering an agentic era, where protein-engineering systems are expected to prioritize future wet-lab experiments rather than merely fit static measurements.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"43rd International Conference on Machine Learning (ICML 2026)","evidence":"Accepted at the 43rd International Conference on Machine Learning (ICML 2026). Data: https://huggingface.co/datasets/JinGao/TadABench-1M . Code: https://github.com/shiyegao/TadABench-1M","evidenceUrl":"https://arxiv.org/abs/2606.02624","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"43rd International Conference on Machine Learning (ICML 2026)","reviewStatus":"accepted","decisionRaw":"Accepted at the 43rd International Conference on Machine Learning (ICML 2026). Data: https://huggingface.co/datasets/JinGao/TadABench-1M . Code: https://github.com/shiyegao/TadABench-1M","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.02624","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at the 43rd International Conference on Machine Learning (ICML 2026). Data: https://huggingface.co/datasets/JinGao/TadABench-1M . Code: https://github.com/shiyegao/TadABench-1M","level":"author-claim"}]}],"publishers":[{"name":"Shanghai Jiao Tong University","organizationType":"academic-lab","sourceUrl":"https://github.com/shiyegao/TadABench-1M","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_taf-med_ece2193a","familyId":"bmf_437ad9931b00","name":"TAF-MED","oneLine":"TAF-MED is a physician-reviewed benchmark of 500 fixed three-turn medical safety scenarios, evaluating LLM responses for unsafe guidance across multi-turn dialogues.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.10258","pdf":"https://arxiv.org/pdf/2608.10258","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.10258"},"evidence":{"snippet":"We introduce TAF-MED, a physician-reviewed benchmark of 500 fixed three-turn scenarios, and evaluate eight LLMs across 4,000 conversations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10258"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"TAF-MED is a physician-reviewed benchmark of 500 fixed three-turn medical safety scenarios, evaluating LLM responses for unsafe guidance across multi-turn dialogues.","whyItMatters":"Addresses the evaluation gap where first-turn safety is an incomplete proxy for conversational safety persistence, providing a protocol for assessing model behavior across complete dialogue trajectories in medical contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c67df0ae4961ae1cee8f27b4ea1848b457e8ef6298a8975347d93513da9dab4a"},"motivation":"Large language models (LLMs) increasingly provide conversational health information that may influence treatment decisions, yet existing benchmarks do not isolate whether medication-safety boundaries persist across follow-ups after explicit self-treatment intent.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10258","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_tailor-bench_613b036f","familyId":"bmf_a1f4ed1db648","name":"Tailor-Bench","oneLine":"Tailor-Bench evaluates visual world models on simulating irregular physical interactions with three scenario modes (regular, unconventional, impossible) and predictive/descriptive generation settings.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.24256","pdf":"https://arxiv.org/pdf/2606.24256","project":null,"code":"https://github.com/tailor-bench/code","data":null,"hfPaper":"https://huggingface.co/papers/2606.24256"},"evidence":{"snippet":"In this work, we introduce Tailor-Bench, a benchmark that challenges world models to simulate irregular physical interactions.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":43,"hfDailySubmittedAt":"2026-06-30T00:00:00.000Z","githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.24256"},"ranking":{"90d":{"score":30,"rank":240,"coverage":0.7,"confidence":"Medium"}},"description":"Tailor-Bench evaluates visual world models on simulating irregular physical interactions with three scenario modes (regular, unconventional, impossible) and predictive/descriptive generation settings.","whyItMatters":"Current benchmarks focus on common interactions; Tailor-Bench exposes the long-tail gap in physical world modeling, testing generalization and constraint awareness beyond typical scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c52a5c3176cb1aaeda5789a3e911165d98cebdc86bb78e307c888f7694d5ceb4"},"motivation":"Physical interactions follow a long-tailed distribution: a set of common and regular interactions dominates human experience and visual data, while a broad spectrum of rare and irregular interactions remains underrepresented.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.24256","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_tangpoetrybench_c91f571c","familyId":"bmf_f51483afb01a","name":"TangPoetryBench","oneLine":"TangPoetryBench evaluates text-to-image models on illustrating classical Chinese Tang poems across ten human-annotated dimensions, with a rubric-conditioned evaluator (PAE).","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.11452","pdf":"https://arxiv.org/pdf/2608.11452","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.11452"},"evidence":{"snippet":"We introduce TangPoetryBench, a multi-dimensional benchmark of 1,280 images (320 classical Chinese Tang poems x 4 state-of-the-art T2I models) with quality-controlled human annotations across ten dimensions.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.11452"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TangPoetryBench evaluates text-to-image models on illustrating classical Chinese Tang poems across ten human-annotated dimensions, with a rubric-conditioned evaluator (PAE).","whyItMatters":"Existing metrics fail to capture cultural and emotional fidelity in poetry-to-image generation; this benchmark provides a multi-dimensional human-annotated dataset and an automated evaluator to support model comparison in this niche domain.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cc83960980390597ab354e5ef5763aa8cfbb047a2d37467427ec2c1d82000353"},"motivation":"Text-to-image (T2I) models are increasingly asked to illustrate literary and cultural content, yet we cannot measure how well an image renders the meaning of a poem.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.11452","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_tar-bench_36cf86db","familyId":"bmf_484abfae69f2","name":"TAR-Bench","oneLine":"Evaluates video-language models on ten traffic anomaly reasoning tasks using 960 human-curated test annotations over 80 held-out clips.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Inspectable","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.10317","pdf":"https://arxiv.org/pdf/2608.10317","project":null,"code":null,"data":"https://huggingface.co/datasets/nvidia/PhysicalAI-Traffic-Anomaly-Reasoning","hfPaper":null},"evidence":{"snippet":"We present TAR (Traffic Anomaly Reasoning) and TAR-Bench datasets, resources for training and evaluating video-language models beyond anomaly detection.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":511,"hfDatasetLikes":20},"source":{"type":"arxiv","id":"2608.10317"},"ranking":{"30d":{"score":52,"rank":20,"coverage":0.15,"confidence":"Low","datasetDownloadRank":8,"datasetRankPopulation":30},"90d":{"score":50,"rank":62,"coverage":0.3,"confidence":"Low","datasetDownloadRank":21,"datasetRankPopulation":66}},"description":"Evaluates video-language models on ten traffic anomaly reasoning tasks using 960 human-curated test annotations over 80 held-out clips.","whyItMatters":"Fills the gap between anomaly detection and higher-level reasoning by testing temporal localization, causal understanding, and multi-task performance.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"5d3718b514b69f7c93ac1d2b4eaa976bd9ca928047209fce9205b19de74448b3"},"motivation":"We present TAR (Traffic Anomaly Reasoning) and TAR-Bench datasets, resources for training and evaluating video-language models beyond anomaly detection.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"The benchmark has a defined evaluation set, scoring via aggregate score, public dataset with download script, and is used in an official challenge.","canonicalNameSource":"abstract","canonicalNameEvidence":"Its evaluation component, TAR-Bench, contains 960 human-curated test annotations for 80 held-out clips"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10317","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":70,"confidence":"Medium","horizon":"7d","reason":"The benchmark is tied to AI City Challenge 2026, has a public Hugging Face dataset, and addresses a practical traffic reasoning use case with broad model evaluation results."},"evaluationMode":"score_submission","publishers":[{"name":"NVIDIA","organizationType":"company-research-lab","sourceUrl":"https://huggingface.co/datasets/nvidia/PhysicalAI-Traffic-Anomaly-Reasoning","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_718d317fbe64cca2","familyId":"catalog_family_718d317fbe64cca2","name":"TAU-bench Airline","oneLine":"Part of τ-bench (TAU-bench), a benchmark for Tool-Agent-User interaction in real-world domains. The airline domain evaluates language agents' ability to interact with users through dynamic conversations while following domain-specific rules and using API tools. Agents must handle airline-related tasks and policies reliably.","description":"Part of τ-bench (TAU-bench), a benchmark for Tool-Agent-User interaction in real-world domains. The airline domain evaluates language agents' ability to interact with users through dynamic conversations while following domain-specific rules and using API tools. Agents must handle airline-related tasks and policies reliably.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Communication","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tau-bench-airline","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_718d317fbe64cca2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tau-bench-airline"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tau-bench-airline","url":"https://llm-stats.com/benchmarks/tau-bench-airline","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","communication","tool calling"],"catalogModelCount":23,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_6c5416dff875789d","familyId":"catalog_family_6c5416dff875789d","name":"TAU-bench Retail","oneLine":"A benchmark for evaluating tool-agent-user interaction in retail environments. Tests language agents' ability to handle dynamic conversations with users while using domain-specific API tools and following policy guidelines. Evaluates agents on tasks like order cancellations, address changes, and order status checks through multi-turn conversations.","description":"A benchmark for evaluating tool-agent-user interaction in retail environments. Tests language agents' ability to handle dynamic conversations with users while using domain-specific API tools and following policy guidelines. Evaluates agents on tasks like order cancellations, address changes, and order status checks through multi-turn conversations.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Communication","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tau-bench-retail","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6c5416dff875789d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tau-bench-retail"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tau-bench-retail","url":"https://llm-stats.com/benchmarks/tau-bench-retail","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","communication","tool calling"],"catalogModelCount":25,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_af6c4c125898ac6b","familyId":"catalog_family_af6c4c125898ac6b","name":"Tau2 Airline","oneLine":"TAU2 airline domain benchmark for evaluating conversational agents in dual-control environments where both AI agents and users interact with tools in airline customer service scenarios. Tests agent coordination, communication, and ability to guide user actions in tasks like flight booking, modifications, cancellations, and refunds.","description":"TAU2 airline domain benchmark for evaluating conversational agents in dual-control environments where both AI agents and users interact with tools in airline customer service scenarios. Tests agent coordination, communication, and ability to guide user actions in tasks like flight booking, modifications, cancellations, and refunds.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Communication","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tau2-airline","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_af6c4c125898ac6b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tau2-airline"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tau2-airline","url":"https://llm-stats.com/benchmarks/tau2-airline","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","communication","tool calling"],"catalogModelCount":24,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_93cc12e0807d1b69","familyId":"catalog_family_93cc12e0807d1b69","name":"Tau2 Retail","oneLine":"τ²-bench retail domain evaluates conversational AI agents in customer service scenarios within a dual-control environment where both agent and user can interact with tools. Tests tool-agent-user interaction, rule adherence, and task consistency in retail customer support contexts.","description":"τ²-bench retail domain evaluates conversational AI agents in customer service scenarios within a dual-control environment where both agent and user can interact with tools. Tests tool-agent-user interaction, rule adherence, and task consistency in retail customer support contexts.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Communication","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tau2-retail","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_93cc12e0807d1b69"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tau2-retail"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tau2-retail","url":"https://llm-stats.com/benchmarks/tau2-retail","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","communication","tool calling"],"catalogModelCount":27,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_3fc20a133285456f","familyId":"catalog_family_3fc20a133285456f","name":"Tau2 Telecom","oneLine":"τ²-Bench telecom domain evaluates conversational agents in a dual-control environment modeled as a Dec-POMDP, where both agent and user use tools in shared telecommunications troubleshooting scenarios that test coordination and communication capabilities.","description":"τ²-Bench telecom domain evaluates conversational agents in a dual-control environment modeled as a Dec-POMDP, where both agent and user use tools in shared telecommunications troubleshooting scenarios that test coordination and communication capabilities.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Communication","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tau2-telecom","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3fc20a133285456f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tau2-telecom"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tau2-telecom","url":"https://llm-stats.com/benchmarks/tau2-telecom","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","communication","tool calling"],"catalogModelCount":36,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_c928f909866afb23","familyId":"catalog_family_c928f909866afb23","name":"Tau3 Airline","oneLine":"τ³-Bench airline domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated airline booking and reservations environment.","description":"τ³-Bench airline domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated airline booking and reservations environment.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tau3-airline","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c928f909866afb23"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tau3-airline"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tau3-airline","url":"https://llm-stats.com/benchmarks/tau3-airline","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","tool calling"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_ebd77eaccb2d583a","familyId":"catalog_family_ebd77eaccb2d583a","name":"Tau3 Banking","oneLine":"τ³-Bench banking domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated retail banking environment.","description":"τ³-Bench banking domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated retail banking environment.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tau3-banking","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ebd77eaccb2d583a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tau3-banking"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tau3-banking","url":"https://llm-stats.com/benchmarks/tau3-banking","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","tool calling"],"catalogModelCount":7,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_6c8193c29e3e5937","familyId":"catalog_family_6c8193c29e3e5937","name":"Tau3 Retail","oneLine":"τ³-Bench retail domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated online retail environment.","description":"τ³-Bench retail domain evaluates agentic models on multi-turn, tool-using customer-support scenarios in a simulated online retail environment.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tau3-retail","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6c8193c29e3e5937"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tau3-retail"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tau3-retail","url":"https://llm-stats.com/benchmarks/tau3-retail","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","tool calling"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_81a8751700459174","familyId":"catalog_family_81a8751700459174","name":"Tau3 Telecom","oneLine":"τ³-Bench telecom domain evaluates agentic models on multi-turn, tool-using customer-support and troubleshooting scenarios in a simulated telecommunications environment.","description":"τ³-Bench telecom domain evaluates agentic models on multi-turn, tool-using customer-support and troubleshooting scenarios in a simulated telecommunications environment.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Agents","Communication","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tau3-telecom","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_81a8751700459174"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tau3-telecom"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tau3-telecom","url":"https://llm-stats.com/benchmarks/tau3-telecom","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","communication","tool calling"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_b5356cb55eb564be","familyId":"catalog_family_b5356cb55eb564be","name":"TAU3-Bench","oneLine":"TAU3-Bench is a benchmark for evaluating general-purpose agent capabilities, testing models on multi-turn interactions with simulated user models, retrieval, and complex decision-making scenarios.","description":"TAU3-Bench is a benchmark for evaluating general-purpose agent capabilities, testing models on multi-turn interactions with simulated user models, retrieval, and complex decision-making scenarios.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Reasoning","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tau3-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b5356cb55eb564be"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tau3-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tau3-bench","url":"https://llm-stats.com/benchmarks/tau3-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","tool calling"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_f867501a711640f0","familyId":"catalog_family_f867501a711640f0","name":"TaxEval v2","oneLine":"A Vals-created set of questions and responses to tax questions","description":"A Vals-created set of questions and responses to tax questions","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/tax_eval_v2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f867501a711640f0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/taxevalv2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"taxEvalV2","url":"https://benchlm.ai/benchmarks/taxevalv2","paperUrl":"https://www.vals.ai/benchmarks/tax_eval_v2","year":"2026","fullName":"Vals TaxEval v2","format":"Accuracy score","tasks":"Tax question answering and response evaluation","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_tc-bench_6009a2aa","familyId":"bmf_4b558e2721e0","name":"TC-Bench","oneLine":"A benchmark dataset for tropical cyclone research with an automated construction pipeline, used to probe scientific alignment of vision foundation models.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24782","pdf":"https://arxiv.org/pdf/2605.24782","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24782"},"evidence":{"snippet":"To operationalize this framework, we release TC-Bench, a global, reproducible benchmark dataset with an automated construction pipeline for tropical cyclone research, and show that current VFMs rely on visual shortcuts that collapse in intense regimes, indicating that scientific alignment does not arise as a natural byproduct of scaling alone.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24782"},"ranking":{},"description":"A benchmark dataset for tropical cyclone research with an automated construction pipeline, used to probe scientific alignment of vision foundation models.","whyItMatters":"Examines whether models rely on visual shortcuts rather than structural invariants in scientific domains.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"488472c69ff7d173ad64263cf6b3f92eae30f62ad5092da7cfa42c4932096a96"},"motivation":"While Vision Foundation Models (VFMs) excel at predictive tasks on satellite imagery, their performance can arise from visual correlations rather than underlying structural invariants, making even perception-based out-of-distribution accuracy a poor proxy for scientific utility.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026","evidence":"Accepted at ICML 2026","evidenceUrl":"https://arxiv.org/abs/2605.24782","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026","reviewStatus":"accepted","decisionRaw":"Accepted at ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2605.24782","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at ICML 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_tca-bench_07c28360","familyId":"bmf_82f0e5522516","name":"TCA-Bench","oneLine":"TCA-Bench is a diagnostic benchmark for evaluating audiovisual binding and temporal relational reasoning in video captioning models, using a decoupled evaluation protocol.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Safety","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.01667","pdf":"https://arxiv.org/pdf/2607.01667","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.01667"},"evidence":{"snippet":"Furthermore, we present TCA-Bench, a diagnostic benchmark utilizing a Decoupled Evaluation Protocol to isolate and quantify model proficiency in audiovisual binding and temporal relational reasoning.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01667"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TCA-Bench is a diagnostic benchmark for evaluating audiovisual binding and temporal relational reasoning in video captioning models, using a decoupled evaluation protocol.","whyItMatters":"It addresses the need for fine-grained evaluation of temporal and cross-modal alignment in audiovisual captioning, offering a protocol to isolate specific model capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0aabbac870ed441d764414f0936488bb2739d0a7f1cec44d73dbfb6e9923b736"},"motivation":"While Multimodal Large Language Models (MLLMs) have advanced video understanding, achieving precise temporal and cross-modal alignment in audiovisual video captioning remains a formidable challenge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01667","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_tcr-bench_7bc5c915","familyId":"bmf_567a72f9e197","name":"TCR-Bench","oneLine":"TCR-Bench is a diagnostic benchmark for table content-level answerability in RAG, focusing on sibling tables with similar schemas but content differences. It evaluates dense retrievers' ability to identify answerable tables.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-20","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.17742","pdf":"https://arxiv.org/pdf/2607.17742","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.17742"},"evidence":{"snippet":"To study this, we introduce TCR-Bench, a diagnostic benchmark for Table Content-level Answerability in RAG, built around sibling tables, i.e., tables with highly similar schemas but subtle content differences.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.17742"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TCR-Bench is a diagnostic benchmark for table content-level answerability in RAG, focusing on sibling tables with similar schemas but content differences. It evaluates dense retrievers' ability to identify answerable tables.","whyItMatters":"This benchmark probes the semantic-answerability gap in retrieval for table RAG, showing that semantic relevance does not guarantee answerability. It provides diagnostic insights into retriever limitations and the potential for answerability-aware reranking, but is primarily for research findings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8bfe57dfb5b78515896c5d95fdbf9fc687d5bd15576a08d6f520c55d592792f4"},"motivation":"Tables are a critical knowledge source in retrieval-augmented generation (RAG), but a retrieved table may lack sufficient evidence to answer a query, a property we call answerability.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.17742","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_tcs-bench_6efad642","familyId":"bmf_4a4db0639414","name":"TCS-BENCH","oneLine":"Evaluates LLMs on research-level theoretical computer science proof generation, using theorem-proving tasks from papers at STOC, FOCS, and SODA. Provides context for self-contained proofs and uses a verification agent to check correctness.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09538","pdf":"https://arxiv.org/pdf/2608.09538","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09538"},"evidence":{"snippet":"We introduce TCS-Bench, a benchmark for evaluating Large Language Models (LLMs) on research-level Theoretical Computer Science (TCS) proof generation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09538"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLMs on research-level theoretical computer science proof generation, using theorem-proving tasks from papers at STOC, FOCS, and SODA. Provides context for self-contained proofs and uses a verification agent to check correctness.","whyItMatters":"Fills a gap in evaluating LLMs on advanced formal reasoning in theoretical computer science, offering a structured protocol with a verification agent that aligns closely with expert judgment, supporting reliable model comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"060db29575e76c4ae56b1dc4f7b4cc70fbeaea26c0107ca4cbda855375d97d94"},"motivation":"We introduce TCS-Bench, a benchmark for evaluating Large Language Models (LLMs) on research-level Theoretical Computer Science (TCS) proof generation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09538","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TCS-Bench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.09538","role":"benchmark-publisher"}],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_telbench_d23c11db","familyId":"bmf_8c4c87f14d99","name":"TELBench","oneLine":"Evaluates span-level error localization in deep-research agent trajectories. TELBench comprises 1,000 instances with annotations of harmful error spans among normal exploration, failed searches, tentative hypotheses, and harmless noise, scored by span-level localization and first-error accuracy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02060","pdf":"https://arxiv.org/pdf/2606.02060","project":null,"code":"https://github.com/NJU-LINK/DRIFT","data":null,"hfPaper":"https://huggingface.co/papers/2606.02060"},"evidence":{"snippet":"From these annotations, we build TELBench, a 1,000-instance benchmark for identifying error spans among normal exploration, failed searches, tentative hypotheses, and harmless noise.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":58,"hfDailySubmittedAt":null,"githubStars":23,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02060"},"ranking":{},"description":"Evaluates span-level error localization in deep-research agent trajectories. TELBench comprises 1,000 instances with annotations of harmful error spans among normal exploration, failed searches, tentative hypotheses, and harmless noise, scored by span-level localization and first-error accuracy.","whyItMatters":"Final-answer evaluation does not reveal which trajectory steps make answers unreliable. This benchmark enables process-level reliability assessment and comparison of error localization methods for deep-research agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ad0404261eeff9c40b323be9628c5f11c01515f0877b2df8dc1b3200e78aa8fb"},"motivation":"Deep-research agents solve tasks through long trajectories of search, tool use, evidence inspection, and answer synthesis.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02060","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"NJU-LINK","organizationType":"academic-lab","sourceUrl":"https://github.com/NJU-LINK/DRIFT","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_telemetrysuffbench_4e3a485c","familyId":"bmf_384333dc77aa","name":"TelemetrySuffBench","oneLine":"TelemetrySuffBench evaluates models on fault-origin diagnosis in agent telemetry, using controlled multi-component traces with delayed-binding faults, paired coarse views, seven-factor telemetry masks, and exact-equal ambiguous origin pairs.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-08","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2608.07899","pdf":"https://arxiv.org/pdf/2608.07899","project":"https://anonymous.4open.science/r/TelemetrySuffBench-E635/README.md","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07899"},"evidence":{"snippet":"We introduce TelemetrySuffBench, a controlled benchmark that separates failure detection, fault-origin localization, and safe abstention under insufficient evidence.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07899"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TelemetrySuffBench evaluates models on fault-origin diagnosis in agent telemetry, using controlled multi-component traces with delayed-binding faults, paired coarse views, seven-factor telemetry masks, and exact-equal ambiguous origin pairs.","whyItMatters":"It addresses the evaluation gap in diagnosing failure origins from agent telemetry, showing that full telemetry yields high localization accuracy but coarse views preserve detection while limiting localization, highlighting the need for explicit decision-to-provenance links and abstention safeguards.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"999e1e8453fe5c641369afe105d68244f23808c3b6efe18cc7cf2ad49a9c3889"},"motivation":"Agent systems increasingly expose execution traces, yet telemetry that reveals a failure may still be inadequate for identifying where that failure originated.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07899","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Anonymous","organizationType":"community","sourceUrl":"https://anonymous.4open.science/r/TelemetrySuffBench-E635/README.md","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_teleswebench_49d08856","familyId":"bmf_ac181644ed2c","name":"TeleSWEBench","oneLine":"TeleSWEBench is a commit-driven benchmark with 734 questions derived from real developer commits in the srsRAN 5G repository. It evaluates LLM-powered software engineering agents in the telecommunications domain using executable unit tests and a hierarchical LLM-as-a-Judge framework across three difficulty tiers.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["Agents","Code"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.05001","pdf":"https://arxiv.org/pdf/2606.05001","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05001"},"evidence":{"snippet":"In this paper, we introduce TeleSWEBench, the first commit-driven benchmark specifically designed to measure an agent's performance in the telecom domain.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05001"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TeleSWEBench is a commit-driven benchmark with 734 questions derived from real developer commits in the srsRAN 5G repository. It evaluates LLM-powered software engineering agents in the telecommunications domain using executable unit tests and a hierarchical LLM-as-a-Judge framework across three difficulty tiers.","whyItMatters":"General-purpose coding benchmarks fail to capture the stateful logic and strict requirements of telecom software, leaving a gap in evaluating ASE tools for this domain. TeleSWEBench provides a domain-specific benchmark with executable tests and a judge framework.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5e92635cd9765bbd9dff8904880eb25ede448fc14a5a22b998e2c055aa3e2ddd"},"motivation":"With the telecommunications field embracing zero touch management alongside novel O-RAN and AI-RAN frameworks, contemporary telecom networks now function as immensely intricate and heavily softwareized codebases.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05001","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_4e2c048ed501de05","familyId":"catalog_family_4e2c048ed501de05","name":"TempCompass","oneLine":"TempCompass is a comprehensive benchmark for evaluating temporal perception capabilities of Video Large Language Models (Video LLMs). It constructs conflicting videos that share identical static content but differ in specific temporal aspects to prevent models from exploiting single-frame bias. The benchmark evaluates multiple temporal aspects including action, motion, speed, temporal order, and attribute changes across diverse task formats including multi-choice QA, yes/no QA, caption matching, and caption generation.","description":"TempCompass is a comprehensive benchmark for evaluating temporal perception capabilities of Video Large Language Models (Video LLMs). It constructs conflicting videos that share identical static content but differ in specific temporal aspects to prevent models from exploiting single-frame bias. The benchmark evaluates multiple temporal aspects including action, motion, speed, temporal order, and attribute changes across diverse task formats including multi-choice QA, yes/no QA, caption matching, and caption generation.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tempcompass","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4e2c048ed501de05"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tempcompass"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tempcompass","url":"https://llm-stats.com/benchmarks/tempcompass","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_tencent-workbuddy-bench_93d020ca","familyId":"bmf_c8a034f326c0","name":"Tencent WorkBuddy Bench","oneLine":"Multi-domain coding-agent benchmark with reverse-engineered tasks across Code, Web, Office, and Security. Each subset has its own scoring instrument; scores are not comparable across subsets.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20911","pdf":"https://arxiv.org/pdf/2607.20911","project":"https://workbuddybench.com/","code":"https://github.com/Tencent/workbuddy-bench","data":"https://huggingface.co/datasets/tencent/workbuddy-bench","hfPaper":"https://huggingface.co/papers/2607.20911"},"evidence":{"snippet":"We introduce Tencent WorkBuddy Bench, a multi-domain evaluation suite for coding agents; this report documents its construction methodology, scoring protocol, and a cross-model leaderboard.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":26,"hfDailySubmittedAt":null,"githubStars":310,"githubScope":"benchmark_repo","hfDatasetDownloads":4176,"hfDatasetLikes":17},"source":{"type":"arxiv","id":"2607.20911"},"ranking":{"90d":{"score":73,"rank":6,"coverage":1.0,"confidence":"High","datasetDownloadRank":6,"datasetRankPopulation":66}},"description":"Multi-domain coding-agent benchmark with reverse-engineered tasks across Code, Web, Office, and Security. Each subset has its own scoring instrument; scores are not comparable across subsets.","whyItMatters":"Addresses contamination in coding-agent evaluation by constructing tasks not discoverable via web search, and provides a reproducible open-source framework for auditable third-party runs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e21fc569d4441078bda1a7ed61d71fc381a4d6b6c9af3cc933eb570a1f43a7f1"},"motivation":"We introduce Tencent WorkBuddy Bench, a multi-domain evaluation suite for coding agents; this report documents its construction methodology, scoring protocol, and a cross-model leaderboard.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20911","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_tensorbench_704e4bd5","familyId":"bmf_a43eac3e43d6","name":"TensorBench","oneLine":"TensorBench is a benchmark of 199 feature-addition and refactoring tasks on an open-source compiler-based tensor framework extending PyTorch. It grades agents by applying patches and running the framework's test suite.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05570","pdf":"https://arxiv.org/pdf/2606.05570","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05570"},"evidence":{"snippet":"We introduce TensorBench, a benchmark of 199 feature-addition and refactoring tasks on an open-source compiler-based tensor framework that extends PyTorch with first-class support for dense and sparse tensors.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05570"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TensorBench is a benchmark of 199 feature-addition and refactoring tasks on an open-source compiler-based tensor framework extending PyTorch. It grades agents by applying patches and running the framework's test suite.","whyItMatters":"Repository-level coding benchmarks face a trade-off between difficulty and evaluation reliability. TensorBench uses automated test-based grading to provide reliable evaluation on challenging tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"db29c09a3706eecfe4e5ad99468e0f373a773a51873b9df3b756d46c06b0939a"},"motivation":"Repository-level coding benchmarks face a trade-off between task difficulty and evaluation reliability: tasks that challenge frontier models often involve large codebases with incomplete test coverage, while human review does not scale.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05570","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"lib_terminal_bench","familyId":"family_terminal_bench","name":"Terminal-Bench","oneLine":"Established benchmark family · Agents & Tool Use.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents & Tool Use"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2025-01-01","releaseDatePrecision":"year","firstRelease":{"year":2025,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2601.11868","pdf":null,"project":"https://www.tbench.ai/","code":"https://github.com/laude-institute/terminal-bench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_terminal_bench"},"ranking":{},"recordType":"family","aliases":["TerminalBench","Terminal-Bench 2.0"],"sourceAttribution":[{"role":"official-project","url":"https://www.tbench.ai/"}],"adoptionRefs":["openai-gpt5","anthropic-claude4"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"versionPolicy":"Keep 2.x as releases of this family unless task semantics materially fork.","capabilityGroups":["Agents","Coding & Software Engineering"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"terminalBench2","url":"https://benchlm.ai/benchmarks/terminal-bench-2","paperUrl":"https://www.tbench.ai/","year":"2026","fullName":"Terminal-Bench 2.0","format":"Interactive CLI agent evaluation","tasks":"Terminal-based software tasks","successorKey":null},{"catalog":"benchlm","sourceId":"terminalBench2","url":"https://benchlm.ai/benchmarks/terminal-bench-2","paperUrl":"https://www.tbench.ai/","year":"2026","fullName":"Terminal-Bench 2.0","format":"Interactive CLI agent evaluation","tasks":"Terminal-based software tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"terminal-bench-2","url":"https://llm-stats.com/benchmarks/terminal-bench-2","datasetSlug":"terminal-bench","versionCount":2,"subsetCount":1,"rowCount":89,"updatedAt":"2026-06-12T15:03:20.735392+00:00","community":true}],"catalogCategories":["coding","agentic","reasoning","agents","code","tool calling"],"catalogModelCount":51,"catalogStarCount":5},{"id":"catalog_08efd551447036bf","familyId":"catalog_family_08efd551447036bf","name":"Terminal-Bench 2.1","oneLine":"Terminal-Bench 2.1 is an updated release of the Terminal-Bench benchmark that tests AI agents' ability to operate a computer via the terminal. It evaluates how well models handle real-world, end-to-end tasks autonomously, including compiling code, training models, setting up servers, system administration, data science workflows, and security tasks.","description":"Terminal-Bench 2.1 is an updated release of the Terminal-Bench benchmark that tests AI agents' ability to operate a computer via the terminal. It evaluates how well models handle real-world, end-to-end tasks autonomously, including compiling code, training models, setting up servers, system administration, data science workflows, and security tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Coding","Agentic","External","Reasoning","Agents","Code","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://api-docs.deepseek.com/zh-cn/updates/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_08efd551447036bf"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/terminalbench21"},{"role":"benchlm","url":"https://benchlm.ai/benchmarks/terminalbench21"},{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsterminalbench21"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/terminal-bench-2.1"}],"catalogSources":[{"catalog":"benchlm","sourceId":"terminalBench21","url":"https://benchlm.ai/benchmarks/terminalbench21","paperUrl":"https://api-docs.deepseek.com/zh-cn/updates/","year":"2026","fullName":"Terminal-Bench 2.1 (provider run)","format":"Interactive task success rate","tasks":"Terminal-based software-agent tasks","successorKey":null},{"catalog":"benchlm","sourceId":"terminalBench21","url":"https://benchlm.ai/benchmarks/terminalbench21","paperUrl":"https://api-docs.deepseek.com/zh-cn/updates/","year":"2026","fullName":"Terminal-Bench 2.1 (provider run)","format":"Interactive task success rate","tasks":"Terminal-based software-agent tasks","successorKey":null},{"catalog":"benchlm","sourceId":"valsTerminalBench21","url":"https://benchlm.ai/benchmarks/valsterminalbench21","paperUrl":"https://www.vals.ai/benchmarks/terminal-bench-2-1","year":"2026","fullName":"Vals Terminal-Bench 2.1","format":"Accuracy score","tasks":"Terminal-based task execution","successorKey":null},{"catalog":"llm-stats","sourceId":"terminal-bench-2.1","url":"https://llm-stats.com/benchmarks/terminal-bench-2.1","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","agentic","external","reasoning","agents","code","tool calling"],"catalogModelCount":30,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_36edf9cc142c1b03","familyId":"catalog_family_36edf9cc142c1b03","name":"Terminal-Bench 3.0","oneLine":"Terminal-Bench 3.0 is a release of the Terminal-Bench benchmark that tests AI agents' ability to operate a computer via the terminal on real-world, end-to-end tasks.","description":"Terminal-Bench 3.0 is a release of the Terminal-Bench benchmark that tests AI agents' ability to operate a computer via the terminal on real-world, end-to-end tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agentic","Reasoning","Agents","Code","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.frontierbench.ai/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_36edf9cc142c1b03"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/terminal-bench-3"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/terminal-bench-3.0"}],"catalogSources":[{"catalog":"benchlm","sourceId":"frontierBench","url":"https://benchlm.ai/benchmarks/terminal-bench-3","paperUrl":"https://www.frontierbench.ai/","year":"2026","fullName":"Terminal-Bench 3.0","format":"Task completion rate","tasks":"74 professional computer-work tasks across 7 domains","successorKey":null},{"catalog":"llm-stats","sourceId":"terminal-bench-3.0","url":"https://llm-stats.com/benchmarks/terminal-bench-3.0","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","agents","code","tool calling"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_e6829d18305cd8cb","familyId":"catalog_family_e6829d18305cd8cb","name":"Terminal-Bench Hard","oneLine":"Terminal-Bench Hard is a harder terminal-agent benchmark variant evaluated with the Terminus-2 harness in Cohere's Command A+ and North Mini Code releases.","description":"Terminal-Bench Hard is a harder terminal-agent benchmark variant evaluated with the Terminus-2 harness in Cohere's Command A+ and North Mini Code releases.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Coding","Reasoning","Agents","Code","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://artificialanalysis.ai/models/grok-4-3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e6829d18305cd8cb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/terminal-bench-hard"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/terminal-bench-hard"}],"catalogSources":[{"catalog":"benchlm","sourceId":"terminalBenchHard","url":"https://benchlm.ai/benchmarks/terminal-bench-hard","paperUrl":"https://artificialanalysis.ai/models/grok-4-3","year":"2026","fullName":"Terminal-Bench Hard","format":"Task success rate","tasks":"Agentic coding and terminal tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"terminal-bench-hard","url":"https://llm-stats.com/benchmarks/terminal-bench-hard","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","reasoning","agents","code","tool calling"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_839250d9e1eaa9d1","familyId":"catalog_family_839250d9e1eaa9d1","name":"Terminus","oneLine":"Terminal-Bench is a benchmark for testing AI agents in real terminal environments, evaluating how well agents can handle real-world, end-to-end tasks autonomously. The benchmark includes tasks spanning coding, system administration, security, data science, model training, file operations, version control, and web development. Terminus is the neutral test-bed agent designed to work with Terminal-Bench, operating purely through tmux sessions without dedicated tools.","description":"Terminal-Bench is a benchmark for testing AI agents in real terminal environments, evaluating how well agents can handle real-world, end-to-end tasks autonomously. The benchmark includes tasks spanning coding, system administration, security, data science, model training, file operations, version control, and web development. Terminus is the neutral test-bed agent designed to work with Terminal-Bench, operating purely through tmux sessions without dedicated tools.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/terminus","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_839250d9e1eaa9d1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/terminus"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"terminus","url":"https://llm-stats.com/benchmarks/terminus","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_terrabench_679be7e3","familyId":"bmf_b1de429d892f","name":"TerraBench","oneLine":"TerraBench is a benchmark for grounded Earth-science reasoning, built on TerraAgent, a ReAct-style framework that couples LLM planning with scientific tools. It includes 403 tasks across three tracks and eight domains with 24,500 verified steps.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Planning","Information retrieval"],"topics":["Reasoning"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13148","pdf":"https://arxiv.org/pdf/2606.13148","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.13148"},"evidence":{"snippet":"We introduce TerraBench, a benchmark for grounded Earth-science reasoning, built on TerraAgent, a ReAct-style executable framework that interleaves reasoning, tool calls, and observations to couple LLM planning with scientific tools for environmental retrieval, geospatial processing, simulation, and artifact-backed computation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13148"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TerraBench is a benchmark for grounded Earth-science reasoning, built on TerraAgent, a ReAct-style framework that couples LLM planning with scientific tools. It includes 403 tasks across three tracks and eight domains with 24,500 verified steps.","whyItMatters":"Earth-science workflows require reasoning over heterogeneous data types. TerraBench unifies these capabilities in a single interface, with process-level metrics and tolerance-aware scoring.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"cf0c3dd97e1cf343dfa8a50d16094860427a3ca3115107919789d9da73088f21"},"motivation":"Climate and environmental decision-making increasingly requires reasoning across heterogeneous inputs, including gridded physical data, satellite imagery, geospatial context, and simulator outputs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13148","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"bm_terralogic_3dd73c53","familyId":"bmf_5ec0cc94dc3b","name":"TerraLogic","oneLine":"TerraLogic evaluates hierarchical geospatial reasoning in Earth observation through 545 scenario-driven tasks spanning optical, SAR, and infrared imagery. Tasks include hazard vulnerability assessment, urban heat island analysis, and forest fragmentation dynamics. Evaluation uses tool-augmented agents with verifiable multi-step workflows and scored by step-wise and end-to-end metrics.","area":"Vision & 3D","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-07-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.12497","pdf":"https://arxiv.org/pdf/2607.12497","project":null,"code":"https://github.com/Ireliya/TerraLogic","data":null,"hfPaper":"https://huggingface.co/papers/2607.12497"},"evidence":{"snippet":"To address this gap, we introduce TerraLogic, a benchmark for geospatial reasoning.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":26,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.12497"},"ranking":{"90d":{"score":47,"rank":88,"coverage":0.55,"confidence":"Low"}},"description":"TerraLogic evaluates hierarchical geospatial reasoning in Earth observation through 545 scenario-driven tasks spanning optical, SAR, and infrared imagery. Tasks include hazard vulnerability assessment, urban heat island analysis, and forest fragmentation dynamics. Evaluation uses tool-augmented agents with verifiable multi-step workflows and scored by step-wise and end-to-end metrics.","whyItMatters":"Existing remote sensing benchmarks primarily target perception tasks, leaving a gap in assessing cognitive-level geospatial reasoning. TerraLogic provides a fixed dataset and protocol for comparing agent performance on compositional, long-horizon analysis, enabling systematic evaluation of tool-augmented reasoning across modalities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9139470f269706e477c14f4a154a97756138d19c3ecb1e3455aa4009b2bd61c2"},"motivation":"Beyond perception, reasoning is essential in remote sensing for advanced interpretation, inference, and decision-making.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.12497","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TerraLogic Team","organizationType":"academic-lab","sourceUrl":"https://github.com/Ireliya/TerraLogic","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_testevo-bench_296298e8","familyId":"bmf_b73f50c8cb68","name":"TestEvo-Bench","oneLine":"TestEvo-Bench evaluates test and code co-evolution tasks from real commit histories. It includes two tracks: test generation (write new tests for changed behavior) and test update (adapt failing tests). Tasks are packaged with environment configurations for execution-grounded metrics like pass rate, coverage, and mutation score. The live component uses timestamps to enable post-training-cutoff evaluation.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.02469","pdf":"https://arxiv.org/pdf/2607.02469","project":"https://www.testevo-bench.com","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.02469"},"evidence":{"snippet":"We introduce TestEvo-Bench, a benchmark of test and code co-evolution tasks mined from software repositories, with two tracks: in test generation, the agent shall write new tests to capture the new software behavior; in test update, the agent shall adapt failing existing tests to the changed software behavior.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02469"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TestEvo-Bench evaluates test and code co-evolution tasks from real commit histories. It includes two tracks: test generation (write new tests for changed behavior) and test update (adapt failing tests). Tasks are packaged with environment configurations for execution-grounded metrics like pass rate, coverage, and mutation score. The live component uses timestamps to enable post-training-cutoff evaluation.","whyItMatters":"Existing benchmarks often separate tests from code changes, relying on static metadata without verifying executability. TestEvo-Bench provides a dynamic, execution-grounded evaluation that reflects real-world test maintenance, enabling assessment of test automation agents in tracking code changes. Its live design reduces data leakage risk, making it valuable for comparing agents in practical settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f56c6e3473f4e16d55190c712fc8c8ec2a11b35ad6ca6e4170c335464af40a5d"},"motivation":"Software tests and code evolve together: a code change should be followed by new or updated tests that record the new software behavior.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02469","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_cf80cd8aed482d5d","familyId":"catalog_family_cf80cd8aed482d5d","name":"testing","oneLine":"Catalog-listed benchmark; original-source verification is pending.","description":"Catalog-listed benchmark; original-source verification is pending.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":[],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/community:fd462fc2-283c-4967-bd7d-b39d7c661807","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_cf80cd8aed482d5d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/community:fd462fc2-283c-4967-bd7d-b39d7c661807"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"community:fd462fc2-283c-4967-bd7d-b39d7c661807","url":"https://llm-stats.com/benchmarks/community:fd462fc2-283c-4967-bd7d-b39d7c661807","datasetSlug":"testing","versionCount":1,"subsetCount":1,"rowCount":null,"updatedAt":"2026-02-28T11:12:49.700817+00:00","community":true}],"catalogCategories":[],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_texfix-bench_827bc253","familyId":"bmf_2ca72336c626","name":"TeXFix-Bench","oneLine":"TeXFix-Bench evaluates LLM-based full-source document repair across LaTeX, Typst, and Markdown. It provides 10,437 repair instances derived from 743 openly licensed seeds, with a fixed zero-shot protocol and provider-pinned routing.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Inspectable","releasedAt":"2026-08-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07617","pdf":"https://arxiv.org/pdf/2608.07617","project":"https://doi.org/10.5281/zenodo.21831797","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07617"},"evidence":{"snippet":"We present TeXFix-Bench, a multi-format benchmark for LLM-based full-source document repair grounded in a mined fault taxonomy.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07617"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TeXFix-Bench evaluates LLM-based full-source document repair across LaTeX, Typst, and Markdown. It provides 10,437 repair instances derived from 743 openly licensed seeds, with a fixed zero-shot protocol and provider-pinned routing.","whyItMatters":"The benchmark fills a gap in document-repair evaluation by grounding faults in a mined taxonomy, offering a reproducible protocol to compare models on compile success and content restoration. This supports practical selection of models for document repair tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"acc301d5ec7116053aeaab1da5a6b9174b9ee3c12137db81658a85a793be5a30"},"motivation":"Scientific and technical writing depends on markup sources that must compile: LaTeX, Typst, and Markdown pipelines fail on missing delimiters, mismatched environments, broken imports, or package conflicts.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07617","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Zenodo","organizationType":"community","sourceUrl":"https://doi.org/10.5281/zenodo.21831797","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_textrich_83260e1b","familyId":"bmf_324a507fa21f","name":"TextRich","oneLine":"TextRich evaluates detection of AI-generated text-rich images across six categories including posters, charts, receipts, tables, and UI screenshots, with a released dataset on Hugging Face.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19259","pdf":"https://arxiv.org/pdf/2606.19259","project":null,"code":null,"data":"https://huggingface.co/datasets/Shuyiww/TextRich","hfPaper":"https://huggingface.co/papers/2606.19259"},"evidence":{"snippet":"In this paper, we introduce TextRich, a multi-domain benchmark for detecting text-rich images generated by OpenAI's GPT-Image-2.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":1521,"hfDatasetLikes":1},"source":{"type":"arxiv","id":"2606.19259"},"ranking":{"90d":{"score":46,"rank":102,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":13,"datasetRankPopulation":66}},"description":"TextRich evaluates detection of AI-generated text-rich images across six categories including posters, charts, receipts, tables, and UI screenshots, with a released dataset on Hugging Face.","whyItMatters":"Addresses the gap in detecting synthetic images with heavy text content, which is critical for digital trust and content authenticity.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"24ec9611f382fd1f8df4b61d8cc46247ee3acd4196a02b7087234a566ccefaf6"},"motivation":"Text-rich images often contain privacy-sensitive, transactional, or decision-relevant information.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19259","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_204ea79490a8d1fd","familyId":"catalog_family_204ea79490a8d1fd","name":"TextVQA","oneLine":"TextVQA contains 45,336 questions on 28,408 images that require reasoning about text to answer. Introduced to benchmark VQA models' ability to read and reason about text within images, particularly for assistive technologies for visually impaired users. The dataset addresses the gap where existing VQA datasets had few text-based questions or were too small.","description":"TextVQA contains 45,336 questions on 28,408 images that require reasoning about text to answer. Introduced to benchmark VQA models' ability to read and reason about text within images, particularly for assistive technologies for visually impaired users. The dataset addresses the gap where existing VQA datasets had few text-based questions or were too small.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Image To Text","Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/textvqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_204ea79490a8d1fd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/textvqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"textvqa","url":"https://llm-stats.com/benchmarks/textvqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image to text","multimodal","vision"],"catalogModelCount":16,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_tf-refusalbench_fb205626","familyId":"bmf_c969544e9b3f","name":"TF-RefusalBench","oneLine":"TF-RefusalBench is a multilingual benchmark for criminal-law translation and summarization derived from public Swiss Supreme Court rulings. It contains 5,200 prompts across French, German, Italian, and English, designed to trigger refusals in LLMs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.23375","pdf":"https://arxiv.org/pdf/2606.23375","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.23375"},"evidence":{"snippet":"To measure this phenomenon, we introduce TF-RefusalBench, a multilingual benchmark for criminal-law translation and summarization derived from public Swiss Supreme Court rulings.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.23375"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TF-RefusalBench is a multilingual benchmark for criminal-law translation and summarization derived from public Swiss Supreme Court rulings. It contains 5,200 prompts across French, German, Italian, and English, designed to trigger refusals in LLMs.","whyItMatters":"The benchmark addresses the challenge of evaluating over-alignment in LLMs performing legitimate legal tasks. It provides a standardized way to measure refusal behavior across languages and task types, aiding in the selection and mitigation of models for sensitive translation and summarization work.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"437f0e720580a98c5e3c2868201b51df55e7b88ae14ddf635fca01c055777d3c"},"motivation":"While the wider applicability of LLMs in the legal field is currently debated due to their reliability and the gravity of any errors, narrow uses with well-understood and mitigated risks have emerged.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.23375","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_the-complexity-ceiling-benchmark_2e65280a","familyId":"bmf_b80ea254c0bc","name":"The Complexity Ceiling Benchmark","oneLine":"Complexity Ceiling Benchmark evaluates sequential reasoning decay with depth scaling across grounded spatial state-tracking, symbolic pointer manipulation, and transitive relational inference.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29278","pdf":"https://arxiv.org/pdf/2606.29278","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.29278"},"evidence":{"snippet":"We introduce the Complexity Ceiling Benchmark (CCB), a controlled evaluation of how language-model reasoning decays as the number of required sequential steps grows.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29278"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Complexity Ceiling Benchmark evaluates sequential reasoning decay with depth scaling across grounded spatial state-tracking, symbolic pointer manipulation, and transitive relational inference.","whyItMatters":"It quantifies reasoning degradation with step count, which could inform model development for long-horizon tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a5549ee71f8de4a2195c0119b70c1f425d13303514a26dffb6036ddd896a7fb4"},"motivation":"We introduce the Complexity Ceiling Benchmark (CCB), a controlled evaluation of how language-model reasoning decays as the number of required sequential steps grows.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"1st Workshop on Combining Theory and Benchmarks (CTB), CTB@ICML 2026","evidence":"12 pages, 6 figures. Accepted to the 1st Workshop on Combining Theory and Benchmarks (CTB), CTB@ICML 2026","evidenceUrl":"https://arxiv.org/abs/2606.29278","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"1st Workshop on Combining Theory and Benchmarks (CTB), CTB@ICML 2026","reviewStatus":"accepted","decisionRaw":"12 pages, 6 figures. Accepted to the 1st Workshop on Combining Theory and Benchmarks (CTB), CTB@ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.29278","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"12 pages, 6 figures. Accepted to the 1st Workshop on Combining Theory and Benchmarks (CTB), CTB@ICML 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_the-energy-society_dce5cbc5","familyId":"bmf_ac7dd1dd7b14","name":"The Energy Society","oneLine":"The Energy Society is a multi-agent simulation environment where LLM agents operate under energy constraints tied to token generation, completing jobs and donating energy to survive. The evaluation measures emergent cooperative and competitive behavior across different model sizes and incentive settings.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.14865","pdf":"https://arxiv.org/pdf/2607.14865","project":null,"code":"https://github.com/LucasBergholdt/EnergySociety","data":null,"hfPaper":"https://huggingface.co/papers/2607.14865"},"evidence":{"snippet":"The Energy Society is a compact testbed for studying the interaction between token costs and group incentives under a survival pressure.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14865"},"ranking":{"90d":{"score":28,"rank":274,"coverage":0.55,"confidence":"Low"}},"description":"The Energy Society is a multi-agent simulation environment where LLM agents operate under energy constraints tied to token generation, completing jobs and donating energy to survive. The evaluation measures emergent cooperative and competitive behavior across different model sizes and incentive settings.","whyItMatters":"The environment addresses the gap in studying how energy costs and survival pressures shape agent cooperation and competition, providing a testbed for evaluating multi-agent systems in resource-constrained settings. It could support decisions on agent design for sustainable and cooperative AI systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e4caeb06b30d2dd8147a970256106bc38c4eaf237a4f8b34b837194950d1762f"},"motivation":"LLM-based agents are increasingly deployed in multi-agent environments whose incentives can shape their behavior.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"AITC 2026","evidence":"Accepted at AITC 2026","evidenceUrl":"https://arxiv.org/abs/2607.14865","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"AITC 2026","reviewStatus":"accepted","decisionRaw":"Accepted at AITC 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.14865","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at AITC 2026","level":"author-claim"}]}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_the-image-reconstruction-game_055fc5b9","familyId":"bmf_33f45be8b058","name":"The Image Reconstruction Game","oneLine":"A benchmark for iterative multimodal dialogue where a vision-language model issues corrective instructions to an image generator over multiple turns, with accumulated common ground observable as a rendered image.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01901","pdf":"https://arxiv.org/pdf/2606.01901","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01901"},"evidence":{"snippet":"We introduce the Image Reconstruction Game, a fully automated benchmark in which a vision-language model issues corrective instructions to an image generator across multiple turns, making accumulated common ground directly observable as a rendered image.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01901"},"ranking":{},"description":"A benchmark for iterative multimodal dialogue where a vision-language model issues corrective instructions to an image generator over multiple turns, with accumulated common ground observable as a rendered image.","whyItMatters":"Addresses the evaluation of interactive language-vision systems and common ground building, but lacks a clear public comparison path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0dde71fac170cc5190020b94b0e20a3363f009c360739789a157fe9b9c2bf3ca"},"motivation":"We introduce the Image Reconstruction Game, a fully automated benchmark in which a vision-language model issues corrective instructions to an image generator across multiple turns, making accumulated common ground directly observable as a rendered image.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01901","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_the-imitator-game_8d69f8a7","familyId":"bmf_bb4e72e4fc51","name":"The Imitator Game","oneLine":"Evaluates robot imitation across four levels of increasing divergence between demonstration and robot scene, plus an open platform for blind A/B human evaluation.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.22301v1","pdf":"https://arxiv.org/pdf/2608.22301v1","project":"https://imitator-game.github.io","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce The Imitator Game, a four-level benchmark (L0-L3) that progressively widens the gap between the human demonstration and the robot's own scene, isolating where trajectory replay ceases to suffice and task understanding becomes necessary.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22301"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates robot imitation across four levels of increasing divergence between demonstration and robot scene, plus an open platform for blind A/B human evaluation.","whyItMatters":"Identifies functional substitution as the barrier to intent-level imitation, moving evaluation beyond trajectory replay toward task understanding.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"e1a1260fa9be4332576d407307a796092842484d8fa992aad86139db8bb89eaf"},"motivation":"Humans imitate at the level of intent: given a demonstration, we infer its goal and carry it out with whatever tools, objects, and layouts are at hand.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with project website and described open arena, though no direct artifact access is provided in the input.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce The Imitator Game, a four-level benchmark (L0-L3) that progressively widens the gap"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22301v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T17:37:25.889905Z"},"attentionForecast":{"score":40,"confidence":"Low","horizon":"7d","reason":"Robot imitation benchmark with a large dataset but no code or direct download links supplied, likely limiting immediate adoption and attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_800bf1411706621d","familyId":"catalog_family_800bf1411706621d","name":"The Last Ones completion","oneLine":"Share of runs that completed the 32-step cyber range within the 100-million-token limit.","description":"Share of runs that completed the 32-step cyber range within the 100-million-token limit.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_800bf1411706621d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/lastonescyberrangecompletion"}],"catalogSources":[{"catalog":"benchlm","sourceId":"lastOnesCyberRangeCompletion","url":"https://benchlm.ai/benchmarks/lastonescyberrangecompletion","paperUrl":"https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities","year":"2026","fullName":"The Last Ones Cyber Range Completion Rate","format":"Completion rate","tasks":"10 long-horizon runs","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_f38bd4881ee6dcfe","familyId":"catalog_family_f38bd4881ee6dcfe","name":"The Last Ones steps","oneLine":"Average step reached on a 32-step long-horizon cyber range.","description":"Average step reached on a 32-step long-horizon cyber range.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f38bd4881ee6dcfe"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/lastonescyberrangesteps"}],"catalogSources":[{"catalog":"benchlm","sourceId":"lastOnesCyberRangeSteps","url":"https://benchlm.ai/benchmarks/lastonescyberrangesteps","paperUrl":"https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities","year":"2026","fullName":"The Last Ones Average Progress","format":"Average step reached","tasks":"32-step long-horizon cyber range","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_the-value-engine-benchmark_3b9623cf","familyId":"bmf_7c1e90c7b46d","name":"The Value Engine Benchmark (VEB)","oneLine":"An environment for evaluating and training LLM agents on multi-touch enterprise-sales negotiation, grading episodes with a sales-methodology rubric to produce scalar rewards and structured diagnostics.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/rudycelekli/the-value-engine-benchmark","pdf":null,"project":"https://thevalueengine.ai","code":"https://github.com/rudycelekli/the-value-engine-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"the-value-engine-benchmark VEB: an evidence-graded RL environment and benchmark for LLM agents on multi-touch enterprise-sales negotiation.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:rudycelekli/the-value-engine-benchmark"},"ranking":{"30d":{"score":23,"rank":138,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":342,"coverage":0.55,"confidence":"Low"}},"description":"An environment for evaluating and training LLM agents on multi-touch enterprise-sales negotiation, grading episodes with a sales-methodology rubric to produce scalar rewards and structured diagnostics.","whyItMatters":"Stresses long-horizon planning, value framing, price discipline, and honesty under pressure, providing a methodology-controlled testbed for negotiation agents.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"67ca7eb04b98d01812a77d20c6f9db9284b84aa086e668ca2f064085737cec46"},"motivation":"the-value-engine-benchmark VEB: an evidence-graded RL environment and benchmark for LLM agents on multi-touch enterprise-sales negotiation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/rudycelekli/the-value-engine-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T06:55:51.988821Z"},"attentionForecast":{"score":60,"confidence":"Medium","horizon":"7d","reason":"The benchmark leverages a realistic enterprise negotiation setting with a rigorous evaluation grid, appealing to both RL and LLM agent researchers."},"evaluationMode":"score_submission","publishers":[{"name":"The Value Engine","organizationType":"community","sourceUrl":"https://thevalueengine.ai","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_theorembench_8bd203fd","familyId":"bmf_e8410a03abbd","name":"TheoremBench","oneLine":"Evaluates LLMs on theorem proving in Lean4 using classical theorems, with two versions: main and premised. Includes metrics for theorem-level coverage and token efficiency to assess partial progress and proof structure.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09450","pdf":"https://arxiv.org/pdf/2606.09450","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.09450"},"evidence":{"snippet":"We introduce TheoremBench, a Lean4 benchmark designed to evaluate theorem provers beyond contest settings.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09450"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLMs on theorem proving in Lean4 using classical theorems, with two versions: main and premised. Includes metrics for theorem-level coverage and token efficiency to assess partial progress and proof structure.","whyItMatters":"Provides a more realistic evaluation of provers beyond contest problems, revealing biases toward easy subtheorems and inefficient proof strategies. Supports finer-grained analysis of formal reasoning capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"d903f06021d84a12e5ccc9bb1a83178e23d4f6730d1725d9b4118f5c72bd16ad"},"motivation":"LLMs have recently achieved strong results on formal proving benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09450","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TheoremBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.09450","role":"benchmark-publisher"}],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_f3ee08b150f0c57e","familyId":"catalog_family_f3ee08b150f0c57e","name":"TheoremQA","oneLine":"A theorem-driven question answering dataset containing 800 high-quality questions covering 350+ theorems from Math, Physics, EE&CS, and Finance. Designed to evaluate AI models' capabilities to apply theorems to solve challenging university-level science problems.","description":"A theorem-driven question answering dataset containing 800 high-quality questions covering 350+ theorems from Math, Physics, EE&CS, and Finance. Designed to evaluate AI models' capabilities to apply theorems to solve challenging university-level science problems.","area":"Mathematical Reasoning","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Math","Physics","Reasoning","Finance"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/theoremqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f3ee08b150f0c57e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/theoremqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"theoremqa","url":"https://llm-stats.com/benchmarks/theoremqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","physics","reasoning","finance"],"catalogModelCount":6,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"specific"},{"id":"bm_therapeuticsbench_0f53fc28","familyId":"bmf_d3ec29543e47","name":"TherapeuticsBench","oneLine":"TxBench-PP evaluates AI agents on small-molecule preclinical pharmacology tasks including mechanism-of-action, pharmacodynamics, and safety reasoning, using realistic workflow snapshots and deterministic scoring.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals"],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19245","pdf":"https://arxiv.org/pdf/2606.19245","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.19245"},"evidence":{"snippet":"We introduce TherapeuticsBench Preclinical Pharmacology (TxBench-PP), a verifiable benchmark for small-molecule preclinical pharmacology and the first focused slice of a broader TherapeuticsBench effort across drug-discovery stages and therapeutic modalities.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19245"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TxBench-PP evaluates AI agents on small-molecule preclinical pharmacology tasks including mechanism-of-action, pharmacodynamics, and safety reasoning, using realistic workflow snapshots and deterministic scoring.","whyItMatters":"Provides verifiable evaluation of agents on realistic drug discovery decisions, addressing the need for trusted benchmarks in high-stakes scientific applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5cb027a680373e4904e23518118872d33a37a614d7309b0be9862ec985e40b8b"},"motivation":"Artificial intelligence (AI) agents promise to accelerate drug discovery by compressing interpretation and decision-making loops, but practical deployment requires trusted evaluation on realistic program decisions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19245","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_one-success-isn-t-reliability-thinkingbox-_a377202a","familyId":"bmf_c4ef2d2a7c2c","name":"Thinkingbox-Bench","oneLine":"Evaluates LLM agents on 507 stateful business workflows in an executable sandbox, with task-specific checks that accept valid trajectories and reject wrong, missing, or extra effects.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-20","firstSeenAt":"2026-08-21","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.19741","pdf":"https://arxiv.org/pdf/2608.19741","project":null,"code":"https://github.com/microsoft/thinkingbox","data":null,"hfPaper":null},"evidence":{"snippet":"We release both Thinkingbox and Thinkingbox-Bench: https://github.com/microsoft/thinkingbox","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":9,"hfDailySubmittedAt":"2026-08-25T00:00:00.000Z","githubStars":20,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.19741"},"ranking":{"30d":{"score":72,"rank":27,"coverage":0.85,"confidence":"High"},"90d":{"score":67,"rank":94,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates LLM agents on 507 stateful business workflows in an executable sandbox, with task-specific checks that accept valid trajectories and reject wrong, missing, or extra effects.","whyItMatters":"Moves beyond response-level or tool-call-level evaluation by requiring correct persistent state transitions, addressing a key gap in assessing agents for consequential business tasks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"68bd8f809edbe512ec070ff8dd5c435d086e10f13fe3cbada5b11b243391e171"},"motivation":"Recent agent benchmarks increasingly ground evaluation in executable environments, from code repair to web navigation, app APIs, and function calling.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"The paper introduces a named benchmark with a sandbox framework, explicit evaluation checks, and release of code and data via GitHub, meeting the criteria for a reusable benchmark.","canonicalNameSource":"abstract","canonicalNameEvidence":"Thinkingbox-bench contains 507 policy-conditioned workflows across numerous scenarios"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.19741","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-21T04:30:09.569171Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"The focus on stateful business workflows and agent reliability, combined with Microsoft affiliation and open-source release, likely drives above-average immediate interest."},"evaluationMode":"public_reusable","publishers":[{"name":"Microsoft","organizationType":"company-research-lab","sourceUrl":"https://github.com/microsoft/thinkingbox","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_thorarena_63609675","familyId":"bmf_bc2239e32cf5","name":"ThorArena","oneLine":"Presents a benchmark for force-aware humanoid interaction using demonstrations with synchronized motion and force data. Includes Force-Aware Tracking Score and a simulation protocol.","area":"Robotics & Embodied AI","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.RO"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.06052","pdf":"https://arxiv.org/pdf/2607.06052","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.06052"},"evidence":{"snippet":"In this paper, we present ThorArena, a benchmark for evaluating force-aware humanoid interaction based on human demonstrations with synchronized motion and force measurements.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06052"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Presents a benchmark for force-aware humanoid interaction using demonstrations with synchronized motion and force data. Includes Force-Aware Tracking Score and a simulation protocol.","whyItMatters":"Could fill a gap in evaluating humanoid control under physical interaction, which is often overlooked. But unclear evaluation environment and public availability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"be9d6fd0375cd1a2e313a170ae97922221dc5b75fa5f846353474380a1063886"},"motivation":"Humanoid robots are increasingly expected to perform contact-rich tasks that require not only accurate whole-body motion but also robust physical interaction with surrounding objects and humans.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06052","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"general"},{"id":"bm_thousandworlds_3c4422be","familyId":"bmf_f43b626e717d","name":"ThousandWorlds","oneLine":"ThousandWorlds is a benchmark for climate emulation of potentially habitable exoplanets. It provides a dataset of approximately 1800 simulations from five global climate models, mapping eight planet parameters to 3D atmospheric fields. It includes three nested benchmark subsets: single-simulator regression, multi-simulator regression with complete observations, and multi-simulator regression with structured missingness. Two evaluation protocols are provided: one for ranking methods and one measuring performance relative to inter-model disagreement.","area":"Language & Knowledge","applicationDomains":["Science & Research"],"primaryDomain":"Science & Research","industrySectors":["Materials & Chemicals"],"capabilities":[],"topics":["cs.LG"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18338","pdf":"https://arxiv.org/pdf/2606.18338","project":"https://doi.org/10.57967/hf/8695","code":"https://github.com/edstevenson/ThousandWorlds","data":null,"hfPaper":"https://huggingface.co/papers/2606.18338"},"evidence":{"snippet":"We introduce ThousandWorlds, an ML-ready benchmark for exoclimate emulation and for the broader regime of low-data, multi-simulator, parameter-to-field regression.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":19,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18338"},"ranking":{"90d":{"score":39,"rank":156,"coverage":0.7,"confidence":"Medium"}},"description":"ThousandWorlds is a benchmark for climate emulation of potentially habitable exoplanets. It provides a dataset of approximately 1800 simulations from five global climate models, mapping eight planet parameters to 3D atmospheric fields. It includes three nested benchmark subsets: single-simulator regression, multi-simulator regression with complete observations, and multi-simulator regression with structured missingness. Two evaluation protocols are provided: one for ranking methods and one measuring performance relative to inter-model disagreement.","whyItMatters":"Machine-learning emulators could accelerate exoplanet climate modeling, but progress was limited by the lack of a curated multi-model dataset. This benchmark fills that gap, enabling reproducible comparison of emulation methods and addressing a regime of low-data, multi-simulator regression where deep learning underperforms.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d95e57072fd80935cd0f6c8513412e4ac080412cf8a510c50be8ac015f6d09d5"},"motivation":"The search for life beyond Earth will depend on detecting faint signatures in the atmospheres of potentially habitable exoplanets.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18338","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ThousandWorlds team","organizationType":"academic-lab","sourceUrl":"https://github.com/edstevenson/ThousandWorlds","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_tickingcollabbench_3fac6f85","familyId":"bmf_b34cf6b2cdf3","name":"TickingCollabBench","oneLine":"Evaluates multi-agent collaboration in Minecraft-based tasks with time-sensitive complementary collaboration, requiring agent heterogeneity, mandatory collaboration, and dynamic environments. The framework generates diverse tasks via YAML specifications and filters invalid configurations.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15684","pdf":"https://arxiv.org/pdf/2606.15684","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.15684"},"evidence":{"snippet":"We present TickingCollabBench, a Minecraft-based multi-agent benchmark for a novel class of time-sensitive complementary collaboration tasks.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15684"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates multi-agent collaboration in Minecraft-based tasks with time-sensitive complementary collaboration, requiring agent heterogeneity, mandatory collaboration, and dynamic environments. The framework generates diverse tasks via YAML specifications and filters invalid configurations.","whyItMatters":"Bridges the gap between static multi-agent benchmarks and real-world scenarios needing real-time coordination under partial observability. Provides a feasibility-aware pipeline for reproducible task generation and evaluation of LLM-based agents in dynamic settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"4f0a9c0e7a5b720b18a5aec27c86c5305ad5aa7ddbbf168a822d07409896ddd7"},"motivation":"We present TickingCollabBench, a Minecraft-based multi-agent benchmark for a novel class of time-sensitive complementary collaboration tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15684","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TickingCollabBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.15684","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_1427274ec26542b9","familyId":"catalog_family_1427274ec26542b9","name":"Time Horizon Index: KSP","oneLine":"A Vals AI agent benchmark that gives each system five days to build and run a space program in Kerbal Space Program.","description":"A Vals AI agent benchmark that gives each system five days to build and run a space program in Kerbal Space Program.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/time_horizon_index","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1427274ec26542b9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valstimehorizonksp"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsTimeHorizonKsp","url":"https://benchlm.ai/benchmarks/valstimehorizonksp","paperUrl":"https://www.vals.ai/benchmarks/time_horizon_index","year":"2026","fullName":"Vals Time Horizon Index: Kerbal Space Program","format":"Mission-ladder progress with partial credit","tasks":"30 progressively harder Kerbal Space Program missions","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_how-good-are-foundation-models-in-longitud_4e643d29","familyId":"bmf_dcb7036bfa01","name":"Time-Aware Multi-View MRI Benchmark","oneLine":"The benchmark comprises 3,920 expert-verified QA pairs from 890 patients across longitudinal MRI timepoints, evaluating temporal reasoning, disease progression, structured localization, sequence ordering, and change localization.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.13309","pdf":"https://arxiv.org/pdf/2608.13309","project":null,"code":"https://github.com/wafaAlghallabi/Time-Aware-MRI","data":null,"hfPaper":null},"evidence":{"snippet":"We introduce the Time-Aware Multi-View MRI Benchmark, an evaluation framework unifying multi-view anatomical input, temporal reasoning across longitudinal scans, and structured localization guidance.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13309"},"ranking":{"30d":{"score":28,"rank":96,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":266,"coverage":0.55,"confidence":"Low"}},"description":"The benchmark comprises 3,920 expert-verified QA pairs from 890 patients across longitudinal MRI timepoints, evaluating temporal reasoning, disease progression, structured localization, sequence ordering, and change localization.","whyItMatters":"Addresses the gap in evaluating vision-language models on longitudinal, multi-view MRI reasoning, which is crucial for clinical progression tracking and treatment assessment.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"252f7b40593a353d851641fd4b878ffb00c82183dfa86be00d967df9773d08c6"},"motivation":"Magnetic Resonance Imaging (MRI) interpretation is fundamental to clinical decision-making, requiring radiologists to integrate multi-view anatomical planes across sequential timepoints while precisely localizing interval changes.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally introduced with a public code repository and promised dataset release on Hugging Face, providing a clear public reuse path.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce the Time-Aware Multi-View MRI Benchmark, an evaluation framework unifying multi-view anatomical input, temporal reasoning across longitudinal scans, and structured localization guidance."},"publication":{"status":"acceptance_claimed","venue":"MICCAI 2026 (Early Accept)","evidence":"Accepted at MICCAI 2026 (Early Accept). 11 pages, 3 figures, 2 tables","evidenceUrl":"https://arxiv.org/abs/2608.13309","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-19T11:05:13.395391Z"},"venueAttempts":[{"venueName":"MICCAI 2026 (Early Accept)","reviewStatus":"accepted","decisionRaw":"Accepted at MICCAI 2026 (Early Accept). 11 pages, 3 figures, 2 tables","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.13309","observedAt":"2026-08-19T11:05:13.395391Z","rawValue":"Accepted at MICCAI 2026 (Early Accept). 11 pages, 3 figures, 2 tables","level":"author-claim"}]}],"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets a clinically relevant gap with extensive data and multiple model evaluations, likely to attract attention from medical AI and vision-language researchers."},"evaluationMode":"public_reusable","publishers":[{"name":"Mohamed bin Zayed University of Artificial Intelligence","organizationType":"academic-lab","sourceUrl":"https://github.com/wafaAlghallabi/Time-Aware-MRI","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_timesage-ev_e6de9c4a","familyId":"bmf_8aaa3b125367","name":"TimeSage-EV","oneLine":"TimeSage-EV evaluates LLM agents on time series analysis tasks in evolving environments, using 60 institutional scenarios across 6 domains with 1,485 scenario-period QA pairs. Agents receive data and reports, with withheld target releases as ground truth, assessing state identification, data summarization, and outlook reasoning.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.14270","pdf":"https://arxiv.org/pdf/2608.14270","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.14270"},"evidence":{"snippet":"We introduce TimeSage-EV, a live benchmark for agentic time series analysis in evolving environments.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14270"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TimeSage-EV evaluates LLM agents on time series analysis tasks in evolving environments, using 60 institutional scenarios across 6 domains with 1,485 scenario-period QA pairs. Agents receive data and reports, with withheld target releases as ground truth, assessing state identification, data summarization, and outlook reasoning.","whyItMatters":"Existing time series QA benchmarks rely on fixed snapshots, but real-world data is released periodically, affecting conclusions. TimeSage-EV fills this gap by evaluating agentic temporal validity and cutoff-aware evidence use, providing a decision-value for deploying LLM agents in high-stakes domains where data updates matter.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"470e405966f82e0b28501a51e4b77551a24e44cf36ec8afb7d78bfb34e129b83"},"motivation":"Time series analysis in high-stakes domains relies on recurring data releases, where new observations can alter the evidence base and the validity of later conclusions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14270","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TimeSage-EV Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.14270","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_timesage-mt_04a07ed7","familyId":"bmf_2c4a9390c0f0","name":"TimeSage-MT","oneLine":"TimeSage-MT evaluates agentic time series reasoning in multi-turn dialogues, covering 240 tasks and 2,680 turns across 8 domains, with a protocol and leaderboard for comparing systems.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01498","pdf":"https://arxiv.org/pdf/2606.01498","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01498"},"evidence":{"snippet":"In this work, we introduce TimeSage-MT, a multi-turn benchmark for agentic time series reasoning with 240 tasks and 2,680 dialogue turns across 8 real-world domains, spanning basic exploration to decision-oriented analysis.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01498"},"ranking":{},"description":"TimeSage-MT evaluates agentic time series reasoning in multi-turn dialogues, covering 240 tasks and 2,680 turns across 8 domains, with a protocol and leaderboard for comparing systems.","whyItMatters":"It fills a gap in benchmarking agentic time series analysis, which differs from single-step tasks, offering a way to assess multi-turn memory, uncertainty handling, and decision-making, with practical value for developing and comparing LLM agents.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8c6f9a4db9e9b65d32eb16ae6ba3c7b366f31a910e1e8ceda803b20674bc94c5"},"motivation":"Time series data inform critical decisions across many real-world domains.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01498","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"TimeSage-MT Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.01498","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_timevista_65f9c096","familyId":"bmf_4f58296bba2d","name":"TimeVista","oneLine":"TimeVista is a benchmark for evaluating time series forecasting using Vision-Language Models (VLMs) as judges, with 5563 samples and rubrics for micro- and macro-level judgments.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.16173","pdf":"https://arxiv.org/pdf/2606.16173","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.16173"},"evidence":{"snippet":"To this end, we introduce TimeVista, a comprehensive VLM-as-a-Judge benchmark comprising 5563 time series samples paired with detailed evaluation rubrics.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16173"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TimeVista is a benchmark for evaluating time series forecasting using Vision-Language Models (VLMs) as judges, with 5563 samples and rubrics for micro- and macro-level judgments.","whyItMatters":"Addresses limitations of point-wise metrics in time series forecasting, offering a human-aligned evaluation approach that could guide model selection and improvement.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bc73f6da273b03dcc167394a747325e842410c81db3c0eb900d934dd52de6652"},"motivation":"High-quality time series forecasting is pivotal for real-world decision-making.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.16173","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_5a8634ec2cdab971","familyId":"catalog_family_5a8634ec2cdab971","name":"TIR-Bench","oneLine":"A tool-calling and multimodal interaction benchmark for testing visual instruction following and execution reliability.","description":"A tool-calling and multimodal interaction benchmark for testing visual instruction following and execution reliability.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Multimodalgrounded","Multimodal","Reasoning","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5a8634ec2cdab971"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/tirbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tir-bench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"tirBench","url":"https://benchlm.ai/benchmarks/tirbench","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"TIR-Bench","format":"Screenshot-grounded task reasoning","tasks":"Visual agent and interface reasoning","successorKey":null},{"catalog":"llm-stats","sourceId":"tir-bench","url":"https://llm-stats.com/benchmarks/tir-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","agents","tool calling"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"bm_tla-bench_4cd98a83","familyId":"bmf_205d79c0e84d","name":"TLA+-Bench","oneLine":"TLA+-Bench evaluates natural-language to TLA+ specification generation by executing each specification in the TLA+ model checker across the full reachable state space. The dataset includes 403 model-checked gold and 897 parse-only silver specifications, with multiple descriptions and difficulty labels.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-26","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.23425","pdf":"https://arxiv.org/pdf/2607.23425","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.23425"},"evidence":{"snippet":"We present TLA$^{+}$-Bench, a dataset and benchmark that grades by execution.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23425"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TLA+-Bench evaluates natural-language to TLA+ specification generation by executing each specification in the TLA+ model checker across the full reachable state space. The dataset includes 403 model-checked gold and 897 parse-only silver specifications, with multiple descriptions and difficulty labels.","whyItMatters":"Prior benchmarks for formal specification generation grade by reference resemblance or parseability, not correctness. TLA+-Bench provides an execution-grounded oracle that measures whether generated specifications satisfy the required properties, offering a more reliable evaluation signal and revealing a range of correctness scores depending on grading choices.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0e83bddca92dc4f43b97bae05b00093ae43a88970802e95a4f9fbca826ea1ccc"},"motivation":"Large language models increasingly write TLA$^{+}$ formal specifications from natural-language descriptions, but progress is hard to measure: existing resources grade by resemblance to a reference or by whether the output parses, neither of which shows correctness.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.23425","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_8c04a4476ebb94ba","familyId":"catalog_family_8c04a4476ebb94ba","name":"TLDR9+ (test)","oneLine":"A large-scale summarization dataset containing over 9 million training instances extracted from Reddit, designed for extreme summarization (generating one-sentence summaries with high compression and abstraction). More than twice larger than previously proposed datasets.","description":"A large-scale summarization dataset containing over 9 million training instances extracted from Reddit, designed for extreme summarization (generating one-sentence summaries with high compression and abstraction). More than twice larger than previously proposed datasets.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Summarization"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tldr9+-(test)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8c04a4476ebb94ba"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tldr9+-(test)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tldr9+-(test)","url":"https://llm-stats.com/benchmarks/tldr9+-(test)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","summarization"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_8f91a84a4972e22a","familyId":"catalog_family_8f91a84a4972e22a","name":"Toloka Arena","oneLine":"An independent agentic-intelligence evaluation from Toloka using private simulated workflows and a pass^5 metric.","description":"An independent agentic-intelligence evaluation from Toloka using private simulated workflows and a pass^5 metric.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://toloka.ai/arena","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8f91a84a4972e22a"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/tolokaarena"}],"catalogSources":[{"catalog":"benchlm","sourceId":"tolokaArena","url":"https://benchlm.ai/benchmarks/tolokaarena","paperUrl":"https://toloka.ai/arena","year":"2026","fullName":"Toloka Arena","format":"pass^5 arena score","tasks":"Private simulated enterprise workflows","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_5ed728c2fa5d767b","familyId":"catalog_family_5ed728c2fa5d767b","name":"TOMATO","oneLine":"TOMATO (Temporal Reasoning Multimodal Evaluation) assesses multimodal models on motion and temporal perception in video, testing understanding of actions, motion, and changes over time.","description":"TOMATO (Temporal Reasoning Multimodal Evaluation) assesses multimodal models on motion and temporal perception in video, testing understanding of actions, motion, and changes over time.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tomato","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5ed728c2fa5d767b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tomato"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tomato","url":"https://llm-stats.com/benchmarks/tomato","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","video","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_toolalignbench_7a0c33d1","familyId":"bmf_95022812d16e","name":"ToolAlignBench","oneLine":"The evaluation object is a set of 128 scenarios across 16 domains for tool-calling LLM agents in regulated industries, assessing conflicts between safety-aligned values and deployment instructions. The task involves processing confidential documents and measuring override behavior.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-15","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.14285","pdf":"https://arxiv.org/pdf/2607.14285","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.14285"},"evidence":{"snippet":"To empirically verify this phenomenon, we build a benchmark of 128 scenarios across 16 domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14285"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"The evaluation object is a set of 128 scenarios across 16 domains for tool-calling LLM agents in regulated industries, assessing conflicts between safety-aligned values and deployment instructions. The task involves processing confidential documents and measuring override behavior.","whyItMatters":"The evaluation gap is the lack of tests for conflicting value systems in agentic tool use. Practical value lies in identifying liability risks and tuning alignment strategies for regulated deployments.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6bad161e55b7b44e9985b91f2d6762d5b3e449566835eeffc74e38c7ee1bbe6d"},"motivation":"Safety alignment in LLMs aims to align models with human values, but which values take precedence when they conflict?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"Pluralistic Alignment Workshop at ICML 2026","evidence":"Accepted to the Pluralistic Alignment Workshop at ICML 2026","evidenceUrl":"https://arxiv.org/abs/2607.14285","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"Pluralistic Alignment Workshop at ICML 2026","reviewStatus":"accepted","decisionRaw":"Accepted to the Pluralistic Alignment Workshop at ICML 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.14285","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to the Pluralistic Alignment Workshop at ICML 2026","level":"author-claim"}]}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_cb3f74ceede2e47b","familyId":"catalog_family_cb3f74ceede2e47b","name":"Toolathlon","oneLine":"Tool Decathlon is a comprehensive benchmark for evaluating AI agents' ability to use multiple tools across diverse task categories. It measures proficiency in tool selection, sequencing, and execution across ten different tool-use scenarios.","description":"Tool Decathlon is a comprehensive benchmark for evaluating AI agents' ability to use multiple tools across diverse task categories. It measures proficiency in tool selection, sequencing, and execution across ten different tool-use scenarios.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Tool use"],"topics":["Agentic","Reasoning","Agents","Tool Calling"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_cb3f74ceede2e47b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/toolathlon"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/toolathlon"}],"catalogSources":[{"catalog":"benchlm","sourceId":"toolathlon","url":"https://benchlm.ai/benchmarks/toolathlon","paperUrl":"https://openai.com/index/introducing-gpt-5-4-mini-and-nano/","year":"2026","fullName":"Toolathlon","format":"Interactive tool-calling evaluation","tasks":"Multi-tool workflows","successorKey":null},{"catalog":"llm-stats","sourceId":"toolathlon","url":"https://llm-stats.com/benchmarks/toolathlon","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","agents","tool calling"],"catalogModelCount":39,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Tool Calling"],"domainScope":"general"},{"id":"catalog_8e939fabd3da2264","familyId":"catalog_family_8e939fabd3da2264","name":"Toolathlon Verified avg. turns","oneLine":"Average assistant turns per Toolathlon Verified trajectory.","description":"Average assistant turns per Toolathlon Verified trajectory.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8e939fabd3da2264"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/toolathlonverifiedavgturns"}],"catalogSources":[{"catalog":"benchlm","sourceId":"toolathlonVerifiedAvgTurns","url":"https://benchlm.ai/benchmarks/toolathlonverifiedavgturns","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Toolathlon Verified average assistant turns","format":"Average trajectory length","tasks":"108 verified real-world tool-use tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_13b89f959b0052d2","familyId":"catalog_family_13b89f959b0052d2","name":"Toolathlon Verified Pass@3","oneLine":"Fraction of Toolathlon Verified tasks solved in at least one of three trials.","description":"Fraction of Toolathlon Verified tasks solved in at least one of three trials.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_13b89f959b0052d2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/toolathlonverifiedpass3"},{"role":"benchlm","url":"https://benchlm.ai/benchmarks/toolathlonverifiedpass3all"}],"catalogSources":[{"catalog":"benchlm","sourceId":"toolathlonVerifiedPass3","url":"https://benchlm.ai/benchmarks/toolathlonverifiedpass3","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Toolathlon Verified Pass@3","format":"Pass@3","tasks":"108 verified real-world tool-use tasks","successorKey":null},{"catalog":"benchlm","sourceId":"toolathlonVerifiedPass3All","url":"https://benchlm.ai/benchmarks/toolathlonverifiedpass3all","paperUrl":"https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf","year":"2026","fullName":"Toolathlon Verified Pass cubed","format":"All-three-trials pass rate","tasks":"108 verified real-world tool-use tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_057064821fef9227","familyId":"catalog_family_057064821fef9227","name":"Toolathlon-Verified","oneLine":"A verified tool-use benchmark variant for completing multi-step workflows with external tools.","description":"A verified tool-use benchmark variant for completing multi-step workflows with external tools.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_057064821fef9227"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/toolathlonverified"}],"catalogSources":[{"catalog":"benchlm","sourceId":"toolathlonVerified","url":"https://benchlm.ai/benchmarks/toolathlonverified","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"Toolathlon-Verified","format":"Interactive tool-use score","tasks":"Verified multi-tool workflows","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_toolbench-x_cb76a568","familyId":"bmf_da7a379f3f02","name":"ToolBench-X","oneLine":"ToolBench-X evaluates tool-using agents on multi-step tasks with executable tools and automatic scoring, in environments that include recoverable reliability hazards such as specification drift, invocation errors, execution failures, output drift, and cross-source conflicts.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.25819","pdf":"https://arxiv.org/pdf/2606.25819","project":null,"code":"https://github.com/Foreverskyou/ToolBench-X","data":null,"hfPaper":"https://huggingface.co/papers/2606.25819"},"evidence":{"snippet":"We introduce ToolBench-X, a benchmark for evaluating agents under recoverable reliability hazards.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.25819"},"ranking":{"90d":{"score":25,"rank":302,"coverage":0.7,"confidence":"Medium"}},"description":"ToolBench-X evaluates tool-using agents on multi-step tasks with executable tools and automatic scoring, in environments that include recoverable reliability hazards such as specification drift, invocation errors, execution failures, output drift, and cross-source conflicts.","whyItMatters":"Existing tool-use benchmarks assume stable tool environments, leaving a gap in evaluating agent performance under realistic unreliability. ToolBench-X provides a way to measure and compare agents' ability to diagnose and recover from tool hazards, which is crucial for practical deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"47490e3bfa5efc4a4f029089bdfb5b7c29fa6ed5d1611883b19f16895ce9da12"},"motivation":"Large language models are increasingly deployed as agents that solve tasks by interacting with external tool environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.25819","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"ToolBench-X Team","organizationType":"academic-lab","sourceUrl":"https://github.com/Foreverskyou/ToolBench-X","role":"benchmark-publisher"}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_toolfailbench_999a8910","familyId":"bmf_432789540d43","name":"ToolFailBench","oneLine":"ToolFailBench evaluates tool-use failures in LLM agents across 1,000 tasks in finance, medicine, law, cybersecurity, and real estate. It labels traces with Tool-Skip, Result-Ignore, Output-Fabrication, and Unnecessary-Tool-Use, using a rule classifier and LLM judges.","area":"Language & Knowledge","applicationDomains":["Cybersecurity","Finance & Economics"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity","Financial Services"],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.04686","pdf":"https://arxiv.org/pdf/2607.04686","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.04686"},"evidence":{"snippet":"We introduce ToolFailBench, a diagnostic benchmark for measuring tool-use failures across 1,000 tasks in finance, medicine, law, cybersecurity, and real estate.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.04686"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ToolFailBench evaluates tool-use failures in LLM agents across 1,000 tasks in finance, medicine, law, cybersecurity, and real estate. It labels traces with Tool-Skip, Result-Ignore, Output-Fabrication, and Unnecessary-Tool-Use, using a rule classifier and LLM judges.","whyItMatters":"Aggregate accuracy hides distinct failure modes in tool use. This benchmark separates models that fail to call tools from those that call but ignore results, enabling targeted diagnosis of agent reliability.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7bd9bd8aa8b7a1e612f0620acdbd7999fe753d02c91af512f02ca6a92f9c527f"},"motivation":"Tool calling is central to modern language model agents, but aggregate benchmark scores often hide where tool use fails.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.04686","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"bm_toolmenubench_fce3386d","familyId":"bmf_e8024a0dd932","name":"ToolMenuBench","oneLine":"ToolMenuBench is a benchmark for evaluating tool-menu filtering strategies in multi-step LLM agents. It varies tool-menu size, distractor type, state-dependent structure, and risk exposure, and reports filter-level and downstream metrics such as task success, tool calls, and token usage.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15508","pdf":"https://arxiv.org/pdf/2606.15508","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.15508"},"evidence":{"snippet":"We introduce ToolMenuBench, a benchmark for evaluating tool-menu construction in multi-step LLM agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15508"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"ToolMenuBench is a benchmark for evaluating tool-menu filtering strategies in multi-step LLM agents. It varies tool-menu size, distractor type, state-dependent structure, and risk exposure, and reports filter-level and downstream metrics such as task success, tool calls, and token usage.","whyItMatters":"ToolMenuBench addresses the gap in evaluating how tool-menu construction affects reliability, efficiency, and risk in tool-augmented agents. It provides a reusable framework for studying the agent-interface problem with controlled settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7b4704731947dee77ddf9a0cbe3f560a97d6c8155d3759292525b60bfe91cbb9"},"motivation":"Tool-augmented large language model agents increasingly operate over large tool libraries, but existing evaluations often focus on whether a model can call a tool correctly rather than how the visible tool menu shapes reliability, efficiency, and safety-relevant risk exposure.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.15508","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_toolrobustbench_a4d79725","familyId":"bmf_6c40e4d9d25b","name":"ToolRobustBench","oneLine":"Evaluates tool-calling agents under perturbations across the tool-use pipeline, attributing failures to selection, grounding, argument binding, and feedback handling.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.23635","pdf":"https://arxiv.org/pdf/2608.23635","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce ToolRobustBench, a stage-wise diagnostic benchmark for tool-calling agents, where a tool-calling agent is an LLM system that selects a tool, supplies structured arguments, and interprets its returned feedback.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23635"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates tool-calling agents under perturbations across the tool-use pipeline, attributing failures to selection, grounding, argument binding, and feedback handling.","whyItMatters":"Provides deterministic, cascade-aware diagnosis of robustness beyond clean tool-calling accuracy.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:10:57.123445Z","inputHash":"e2f08dde5feb2e47a765dd6fe913e893119e3d487ff6fc4057f116f78faf9498"},"motivation":"Large language models (LLMs) rely on tool calling as a fundamental agent capability, enabling them to invoke external systems and complete tasks beyond text generation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:10:57.123445Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally named, introduces a repeatable evaluation with four perturbation families and scoring protocols, and describes 15,456 instances; paper implies public availability.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce ToolRobustBench, a stage-wise diagnostic benchmark for tool-calling agents"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.23635","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":41,"confidence":"Low","horizon":"7d","reason":"The benchmark targets a hot agent-tooling area with 15K instances across 7 models, but a supported public link is not supplied."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_touchsafebench_f153a852","familyId":"bmf_f09b093c7c32","name":"TouchSafeBench","oneLine":"TouchSafeBench is a physics-grounded benchmark with 2,940 simulated episodes for evaluating collision grounding in vision-language models for safe human-robot collaboration.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics","Multimodal"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.31196","pdf":"https://arxiv.org/pdf/2605.31196","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.31196"},"evidence":{"snippet":"We introduce TouchSafeBench, a physics-grounded benchmark for evaluating collision grounding in vision-language models (VLMs).","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.31196"},"ranking":{},"description":"TouchSafeBench is a physics-grounded benchmark with 2,940 simulated episodes for evaluating collision grounding in vision-language models for safe human-robot collaboration.","whyItMatters":"Targets a critical safety capability in embodied AI, but current evidence lacks public availability and scoring details.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ef272e2854a2d8ec836707474afd698a7b5243512e65980a28624fcd5db8f7e9"},"motivation":"Safe human--robot collaboration requires more than visual description: a monitor must determine whether the robot body is safely separated, already colliding with the scene or a person, or about to collide.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.31196","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_toxscreen_d4a7771e","familyId":"bmf_fa2e802a8fcc","name":"ToxScreen","oneLine":"A benchmark of roughly 800 backdoored LLMs across attack objectives, trigger mechanisms, poisoning rates, model scales, and training mechanisms. Evaluates whether a defender can recover a planted trigger given white-box access and behavior of concern.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CR"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.26849","pdf":"https://arxiv.org/pdf/2607.26849","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.26849"},"evidence":{"snippet":"To evaluate whether a defender can recover such a trigger under realistic settings, we release ToxScreen, a benchmark of roughly 800 backdoored models spanning attack objectives, trigger mechanisms, poisoning rates, model scales, and backdoor training mechanisms.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.26849"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark of roughly 800 backdoored LLMs across attack objectives, trigger mechanisms, poisoning rates, model scales, and training mechanisms. Evaluates whether a defender can recover a planted trigger given white-box access and behavior of concern.","whyItMatters":"Backdoor recovery in LLMs is a critical security challenge; this benchmark provides a standardized testbed for comparing trigger-recovery methods under realistic constraints, informing practical defense strategies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"449dfacec19ae637c5913cadd482413d54fabbfb92fb1e6ba5812945d34da3be"},"motivation":"As large language models (LLMs) are deployed in high-stakes domains, adversaries may poison training data to implant backdoors: hidden triggers that covertly manipulate model behavior at inference time.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.26849","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_trace_e2618c65","familyId":"bmf_eafe895eb811","name":"TRACE","oneLine":"TRACE is an evidence-grounded safety evaluation benchmark covering prompts, reasoning traces, and final responses of Large Reasoning Models (LRMs). It includes prompts in two languages, nine risk categories, ten attack strategies, and per-component safety annotations with supporting evidence. Evaluates 18 guardrail models on safety judgment and evidence extraction.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Safety","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.24232","pdf":"https://arxiv.org/pdf/2608.24232","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To address these limitations, we introduce TRACE, an evidence-grounded safety evaluation benchmark that covers the entire LRM inference pipeline: prompts, reasoning traces, and final responses.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24232"},"ranking":{"30d":{"score":39,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"TRACE is an evidence-grounded safety evaluation benchmark covering prompts, reasoning traces, and final responses of Large Reasoning Models (LRMs). It includes prompts in two languages, nine risk categories, ten attack strategies, and per-component safety annotations with supporting evidence. Evaluates 18 guardrail models on safety judgment and evidence extraction.","whyItMatters":"Safety evaluation of LRMs currently focuses on prompts and final responses, ignoring reasoning traces that may contain unsafe content. TRACE provides a reproducible benchmark with evidence annotations, enabling precise evaluation of guardrails and highlighting challenges in trace-level safety judgment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"334c57270ff5753aad73450c83179a58448f704276a05508875567d44b2f8a82"},"motivation":"Large Reasoning Models (LRMs) generate intermediate reasoning traces that may contain unsafe content, even when their final responses appear safe.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark has a clearly defined evaluation object, stable scoring contract, and explicit release statement, though no code or data link is provided in the artifact evidence.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce TRACE, an evidence-grounded safety evaluation benchmark that covers the entire LRM inference pipeline: prompts, reasoning traces, and final responses."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24232","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"Safety-critical and timely topic with detailed benchmark design; likely to interest safety and alignment researchers."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_trace-bench_6cb6dfd0","familyId":"bmf_4335f0096e7f","name":"TRACE-Bench","oneLine":"Evaluates multi-reference image generation by decomposing prompts into atomic operators (Anchor, Disentangle, Apply, Compose) and scoring operator-aligned capabilities and diagnostic failure localization.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-17","firstSeenAt":"2026-08-19","recognitionConfidence":0.75,"links":{"report":"https://arxiv.org/abs/2608.16765","pdf":"https://arxiv.org/pdf/2608.16765","project":"https://amuseum-whr.github.io/TraceBench","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Building on this formulation, we construct TRACE-Bench, comprising approximately 1,600 evaluation cases across slot counts 1--8, built from 631 formula templates and around 4,000 reference images spanning diverse artistic styles and real-world subjects.","reasonCodes":["coined title prefix ending in Bench or Benchmark","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":13,"hfDailySubmittedAt":"2026-08-18T00:00:00.000Z","githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.16765"},"ranking":{"30d":{"score":22,"rank":159,"coverage":0.85,"confidence":"High"},"90d":{"score":22,"rank":384,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates multi-reference image generation by decomposing prompts into atomic operators (Anchor, Disentangle, Apply, Compose) and scoring operator-aligned capabilities and diagnostic failure localization.","whyItMatters":"Provides a capability-oriented protocol that captures combinatorial complexity and enables per-operator diagnostics, addressing the fragmented coverage and limited diagnostic value of task-type-based benchmarks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T23:23:46.758372Z","inputHash":"47f97fa77b1fea62329af86771fac663ce207f1fa3c9b2e0125bbbda4efa020e"},"motivation":"Despite recent advances in unified multimodal models for multi-reference image generation, existing benchmarks remain organized around predefined task types (e.g., \"subject composition\"), which are ill-suited to this combinatorial setting and lead to fragmented coverage, uncontrolled complexity, and little diagnostic value.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T23:23:46.758372Z","model":"deepseek-v4-pro","decisionReason":"TRACE-Bench has a formal name in the title, a project page with evaluation cases, and an operator-aligned scoring protocol; it provides a public path for model comparison and is accepted at ACM MM 2026.","canonicalNameSource":"paper_title","canonicalNameEvidence":"TRACE-Bench: Decomposing and Diagnosing Multi-Reference Image Generation"},"publication":{"status":"acceptance_claimed","venue":"ACM Multimedia 2026 (ACM MM 2026)","evidence":"Accepted to ACM Multimedia 2026 (ACM MM 2026)","evidenceUrl":"https://arxiv.org/abs/2608.16765","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-19T11:05:13.395391Z"},"venueAttempts":[{"venueName":"ACM Multimedia 2026 (ACM MM 2026)","reviewStatus":"accepted","decisionRaw":"Accepted to ACM Multimedia 2026 (ACM MM 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.16765","observedAt":"2026-08-19T11:05:13.395391Z","rawValue":"Accepted to ACM Multimedia 2026 (ACM MM 2026)","level":"author-claim"}]}],"attentionForecast":{"score":68,"confidence":"Medium","reason":"Formal release with project page, large case count, and acceptance at a major multimedia conference suggests moderate early attention in the multimodal generation community.","horizon":"7d"},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_tradeverse_c4be7871","familyId":"bmf_29d08cd7c2a4","name":"TradeVerse","oneLine":"TradeVerse evaluates LLMs on longitudinal political trade negotiation understanding using reconstructed minutes of 1170 WTO meetings across 5 groups and 89 product groups, with three tasks: predicting HS codes, identifying responding countries, and generating final statements.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.06549","pdf":"https://arxiv.org/pdf/2608.06549","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.06549"},"evidence":{"snippet":"We introduce TradeVerse, a benchmark built from the World Trade Organisation (WTO) specific trade concerns, where member states challenge one another and exchange arguments over multiple rounds, sometimes for years.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06549"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TradeVerse evaluates LLMs on longitudinal political trade negotiation understanding using reconstructed minutes of 1170 WTO meetings across 5 groups and 89 product groups, with three tasks: predicting HS codes, identifying responding countries, and generating final statements.","whyItMatters":"It addresses the evaluation gap of LLMs on longitudinal, multi-turn negotiation data, which is common in real-world political and institutional contexts, and provides a challenging test for tracking context and reasoning over extended interactions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ec0dd436bc7ede2a320921441492b672eb6bb3837aa8e5f78a074f216a2089a5"},"motivation":"LLMs are increasingly being applied to tasks involving institutional and political texts, but existing benchmarks evaluate them on isolated documents or single tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06549","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_96b952f2d4104005","familyId":"catalog_family_96b952f2d4104005","name":"Trae Code Gen","oneLine":"Trae Code Gen is a component of Trae Agent Bench that evaluates implementing new functionality across multiple programming languages in containerized, runnable repositories.","description":"Trae Code Gen is a component of Trae Agent Bench that evaluates implementing new functionality across multiple programming languages in containerized, runnable repositories.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/trae-code-gen","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_96b952f2d4104005"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/trae-code-gen"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"trae-code-gen","url":"https://llm-stats.com/benchmarks/trae-code-gen","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","coding"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_ddff600660f07ae2","familyId":"catalog_family_ddff600660f07ae2","name":"Trae Error Fix","oneLine":"Trae Error Fix is a component of Trae Agent Bench that evaluates fixing existing code across multiple programming languages in containerized, runnable repositories.","description":"Trae Error Fix is a component of Trae Agent Bench that evaluates fixing existing code across multiple programming languages in containerized, runnable repositories.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/trae-error-fix","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ddff600660f07ae2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/trae-error-fix"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"trae-error-fix","url":"https://llm-stats.com/benchmarks/trae-error-fix","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","coding"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_c9d70348c2c65bb3","familyId":"catalog_family_c9d70348c2c65bb3","name":"Translation en→Set1 COMET22","oneLine":"COMET-22 is an ensemble machine translation evaluation metric combining a COMET estimator model trained with Direct Assessments and a multitask model that predicts sentence-level scores and word-level OK/BAD tags. It demonstrates improved correlations compared to state-of-the-art metrics and increased robustness to critical errors.","description":"COMET-22 is an ensemble machine translation evaluation metric combining a COMET estimator model trained with Direct Assessments and a multitask model that predicts sentence-level scores and word-level OK/BAD tags. It demonstrates improved correlations compared to state-of-the-art metrics and increased robustness to critical errors.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/translation-en→set1-comet22","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c9d70348c2c65bb3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/translation-en→set1-comet22"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"translation-en→set1-comet22","url":"https://llm-stats.com/benchmarks/translation-en→set1-comet22","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_4f5d3b6b0451dacd","familyId":"catalog_family_4f5d3b6b0451dacd","name":"Translation en→Set1 spBleu","oneLine":"Translation evaluation using spBLEU (SentencePiece BLEU), a BLEU metric computed over text tokenized with a language-agnostic SentencePiece subword model. Introduced in the FLORES-101 evaluation benchmark for low-resource and multilingual machine translation.","description":"Translation evaluation using spBLEU (SentencePiece BLEU), a BLEU metric computed over text tokenized with a language-agnostic SentencePiece subword model. Introduced in the FLORES-101 evaluation benchmark for low-resource and multilingual machine translation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/translation-en→set1-spbleu","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4f5d3b6b0451dacd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/translation-en→set1-spbleu"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"translation-en→set1-spbleu","url":"https://llm-stats.com/benchmarks/translation-en→set1-spbleu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_2c15ad0533b756e6","familyId":"catalog_family_2c15ad0533b756e6","name":"Translation Set1→en COMET22","oneLine":"COMET-22 is a neural machine translation evaluation metric that uses an ensemble of two models: a COMET estimator trained with Direct Assessments and a multitask model that predicts sentence-level scores and word-level OK/BAD tags. It provides improved correlations with human judgments and increased robustness to critical errors compared to previous metrics.","description":"COMET-22 is a neural machine translation evaluation metric that uses an ensemble of two models: a COMET estimator trained with Direct Assessments and a multitask model that predicts sentence-level scores and word-level OK/BAD tags. It provides improved correlations with human judgments and increased robustness to critical errors compared to previous metrics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/translation-set1→en-comet22","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2c15ad0533b756e6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/translation-set1→en-comet22"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"translation-set1→en-comet22","url":"https://llm-stats.com/benchmarks/translation-set1→en-comet22","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_6b67cff18fbba972","familyId":"catalog_family_6b67cff18fbba972","name":"Translation Set1→en spBleu","oneLine":"spBLEU (SentencePiece BLEU) evaluation metric for machine translation quality assessment, using language-agnostic SentencePiece tokenization with BLEU scoring. Part of the FLORES-101 evaluation benchmark for low-resource and multilingual machine translation.","description":"spBLEU (SentencePiece BLEU) evaluation metric for machine translation quality assessment, using language-agnostic SentencePiece tokenization with BLEU scoring. Part of the FLORES-101 evaluation benchmark for low-resource and multilingual machine translation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/translation-set1→en-spbleu","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6b67cff18fbba972"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/translation-set1→en-spbleu"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"translation-set1→en-spbleu","url":"https://llm-stats.com/benchmarks/translation-set1→en-spbleu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_trapsbench_2047e708","familyId":"bmf_bffec585a3f1","name":"TRAPSBench","oneLine":"TRAPSBench is a procedurally generated video benchmark with 1,404 physics pairs to test epistemic restraint in VLMs, using Penalized Epistemic Calibration Score (PECS).","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.13167","pdf":"https://arxiv.org/pdf/2608.13167","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.13167"},"evidence":{"snippet":"We introduce TRAPSBench, a procedurally generated video benchmark of 1,404 matched physics pairs in which a single targeted change renders the outcome undeterminable from the visual evidence.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13167"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TRAPSBench is a procedurally generated video benchmark with 1,404 physics pairs to test epistemic restraint in VLMs, using Penalized Epistemic Calibration Score (PECS).","whyItMatters":"It highlights that VLMs can internally detect when abstention is required but fail to express it, offering a metric for calibration and restraint evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"308917b653506a73ef52456ad12533ea709e909f580053828fc58540113ff1d5"},"motivation":"When visual evidence is occluded or chaotic, models should abstain.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13167","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_traveleval_64e034c5","familyId":"bmf_281eeec52056","name":"TravelEval","oneLine":"TravelEval evaluates LLM-powered travel planning agents in a realistic sandbox with accommodation pricing and intercity transport data, using six dimensions: accuracy, compliance, temporality, spatiality, economy, and utility. It simulates complete plans with API-integrated geographic information and queuing time.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning"],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01046","pdf":"https://arxiv.org/pdf/2606.01046","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01046"},"evidence":{"snippet":"To address this gap, we introduce TravelEval, a realistic and comprehensive benchmark.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01046"},"ranking":{},"description":"TravelEval evaluates LLM-powered travel planning agents in a realistic sandbox with accommodation pricing and intercity transport data, using six dimensions: accuracy, compliance, temporality, spatiality, economy, and utility. It simulates complete plans with API-integrated geographic information and queuing time.","whyItMatters":"Existing travel planning benchmarks overemphasize constraint compliance, lack real-world data coverage, and miss global plan quality. TravelEval offers a multi-dimensional framework to compare agent planning capabilities, aiding model selection for complex travel tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7a91bfc7574eee24a3fb9131e510d2ff6b23b941211c46e5635638b2fdcb83ce"},"motivation":"The development of Large Language Models (LLMs) has significantly improved travel planning applications, yet evaluating such models is limited by existing benchmarks' limitations: 1) overemphasis on constraint compliance, neglecting multi-dimensional qualities like spatio-temporal cost; 2) datasets lacking real-world authenticity and coverage in key areas (e.g., lodging, transport); and 3) isolated daily plan assess…","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"KDD 2026","evidence":"31pages, 8 figures, accepted by KDD 2026","evidenceUrl":"https://arxiv.org/abs/2606.01046","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"KDD 2026","reviewStatus":"accepted","decisionRaw":"31pages, 8 figures, accepted by KDD 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.01046","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"31pages, 8 figures, accepted by KDD 2026","level":"author-claim"}]}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_treat_ece84fbf","familyId":"bmf_a5075550e1f2","name":"TREAT","oneLine":"TREAT evaluates LLMs' ability to recognize theorem identities from equivalence-preserving formula transformations, with 737 identities and 29,480 transformed rows.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07540","pdf":"https://arxiv.org/pdf/2608.07540","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07540"},"evidence":{"snippet":"We introduce TREAT, a benchmark for evaluating whether large language models can recover known theorem identities from equivalence-preserving formula-level transformations.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07540"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TREAT evaluates LLMs' ability to recognize theorem identities from equivalence-preserving formula transformations, with 737 identities and 29,480 transformed rows.","whyItMatters":"It targets representation-robust access to formal knowledge, critical for AI tools interacting with formal systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"858fffa6bbbcf8b0a17836ac548bed4a1083e14912b97c353b2448955ad8e95e"},"motivation":"AI systems increasingly operate between flexible input representations and formal objects used by downstream tools.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"28th International Symposium on Symbolic and Numeric Algorithms for Scientific Computing (SYNASC 2026)","evidence":"Accepted at 28th International Symposium on Symbolic and Numeric Algorithms for Scientific Computing (SYNASC 2026)","evidenceUrl":"https://arxiv.org/abs/2608.07540","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"28th International Symposium on Symbolic and Numeric Algorithms for Scientific Computing (SYNASC 2026)","reviewStatus":"accepted","decisionRaw":"Accepted at 28th International Symposium on Symbolic and Numeric Algorithms for Scientific Computing (SYNASC 2026)","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.07540","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at 28th International Symposium on Symbolic and Numeric Algorithms for Scientific Computing (SYNASC 2026)","level":"author-claim"}]}],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_b4f97d9fcd658e33","familyId":"catalog_family_b4f97d9fcd658e33","name":"TreeBench","oneLine":"TreeBench evaluates visual grounded reasoning, requiring models to localize and reason about fine-grained visual details.","description":"TreeBench evaluates visual grounded reasoning, requiring models to localize and reason about fine-grained visual details.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Spatial Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/treebench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b4f97d9fcd658e33"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/treebench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"treebench","url":"https://llm-stats.com/benchmarks/treebench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","spatial reasoning","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_treeprobe_2c533886","familyId":"bmf_d1c4a8ee6c7a","name":"TreeProbe","oneLine":"TreeProbe is a dataset of 4,719 expert-adjudicated items covering 467 diseases and 10 subtasks across the Tibetan Tree of Medicine framework, used to evaluate cultural bias in LLMs.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Software & Cloud","Pharma & Biotech"],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00640","pdf":"https://arxiv.org/pdf/2608.00640","project":"https://anonymous.4open.science/r/TreeProbe/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00640"},"evidence":{"snippet":"To address this gap, we introduce TreeProbe, the first cultural-bias benchmark organized around the native Tree of Medicine framework in Tibetan medicine.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00640"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TreeProbe is a dataset of 4,719 expert-adjudicated items covering 467 diseases and 10 subtasks across the Tibetan Tree of Medicine framework, used to evaluate cultural bias in LLMs.","whyItMatters":"It addresses the lack of quantitative tools for assessing cultural bias in Tibetan medicine, offering a diagnostic lens for epistemic fairness in medical AI, though it primarily supports the paper's analysis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"42020c58ef3c04408bca30bc12b2681995ac9a59f92675e4bdfd5e04b0790971"},"motivation":"Large language models are increasingly viewed as a potential means of mitigating global health inequities, yet their outputs often reflect dominant high-resource medical traditions and provide limited coverage of traditional medical knowledge systems.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00640","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness","Coding & Software Engineering"],"domainScope":"specific"},{"id":"bm_triggerbench_0852166e","familyId":"bmf_85eecdfdb4f5","name":"TriggerBench","oneLine":"TriggerBench is a benchmark for evaluating prospective memory (PM) in LLMs, spanning five dimensions across daily assistant and professional workflow scenarios, with matched retrospective memory (RM) controls, contrastive variants, and overloaded triggers, measuring proactive recall, false-alarm rate, and attentional robustness.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-22","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.23459","pdf":"https://arxiv.org/pdf/2606.23459","project":null,"code":"https://github.com/KristenZHANG/TriggerBench-Official","data":null,"hfPaper":"https://huggingface.co/papers/2606.23459"},"evidence":{"snippet":"We introduce TriggerBench, a comprehensive PM benchmark spanning five dimensions across both daily assistants and professional workflows.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.23459"},"ranking":{"90d":{"score":15,"rank":403,"coverage":0.7,"confidence":"Medium"}},"description":"TriggerBench is a benchmark for evaluating prospective memory (PM) in LLMs, spanning five dimensions across daily assistant and professional workflow scenarios, with matched retrospective memory (RM) controls, contrastive variants, and overloaded triggers, measuring proactive recall, false-alarm rate, and attentional robustness.","whyItMatters":"Existing LLM evaluations focus on retrospective memory via explicit queries, leaving prospective memory – the ability to spontaneously act on latent constraints – unevaluated. TriggerBench provides a granular measurement of PM capabilities, revealing a precision-recall trade-off, attentional fragility, and a decay with context length that RM does not exhibit, informing deployment decisions for long interactive applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"60a37abf835e8cac58ae5bad8246d90bf5bc54172cfcc4586079dbbfe3f46c38"},"motivation":"While Large Language Models (LLMs) are increasingly deployed in long interactions, existing evaluations focus predominantly on retrospective memory (RM) via explicit queries.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.23459","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"KristenZHANG (GitHub)","organizationType":"community","sourceUrl":"https://github.com/KristenZHANG/TriggerBench-Official","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d60b25609f91cc15","familyId":"catalog_family_d60b25609f91cc15","name":"TriviaQA","oneLine":"A large-scale reading comprehension dataset containing over 650K question-answer-evidence triples. TriviaQA includes 95K question-answer pairs authored by trivia enthusiasts and independently gathered evidence documents (six per question on average) that provide high quality distant supervision for answering the questions. The dataset features relatively complex, compositional questions with considerable syntactic and lexical variability, requiring cross-sentence reasoning to find answers.","description":"A large-scale reading comprehension dataset containing over 650K question-answer-evidence triples. TriviaQA includes 95K question-answer pairs authored by trivia enthusiasts and independently gathered evidence documents (six per question on average) that provide high quality distant supervision for answering the questions. The dataset features relatively complex, compositional questions with considerable syntactic and lexical variability, requiring cross-sentence reasoning to find answers.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Reasoning","General"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d60b25609f91cc15"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/triviaqa"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/triviaqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"triviaQa","url":"https://benchlm.ai/benchmarks/triviaqa","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"TriviaQA","format":"Exact match","tasks":"Trivia and reading-comprehension QA","successorKey":null},{"catalog":"llm-stats","sourceId":"triviaqa","url":"https://llm-stats.com/benchmarks/triviaqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","reasoning","general"],"catalogModelCount":18,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_triviewbench_b0684034","familyId":"bmf_75312dbd377d","name":"TriViewBench","oneLine":"TriViewBench evaluates multimodal LLMs on multi-view structural reasoning using synthetic 3D scenes with controlled object count and occlusion. It comprises 1,923 scenes and over 14,000 QA pairs across four complexity levels and three reasoning categories: Local Decision, Object Counting, and Global Recovery.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Geometric reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26029","pdf":"https://arxiv.org/pdf/2606.26029","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.26029"},"evidence":{"snippet":"We introduce TriViewBench, a controlled three-view visual reasoning benchmark constructed from synthetic 3D scenes with explicitly parameterized object count and occlusion.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26029"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TriViewBench evaluates multimodal LLMs on multi-view structural reasoning using synthetic 3D scenes with controlled object count and occlusion. It comprises 1,923 scenes and over 14,000 QA pairs across four complexity levels and three reasoning categories: Local Decision, Object Counting, and Global Recovery.","whyItMatters":"TriViewBench provides controlled complexity scaling to isolate structural reasoning capabilities in MLLMs, revealing distinct failure modes and bottlenecks. It enables systematic comparison of models on multi-view spatial reasoning, informing targeted improvements.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c359dc6ba00b933cad70c9512f2674719f6afa83b31e3a1712fbd3a0882a4e02"},"motivation":"Multimodal Large Language Models (MLLMs) demonstrate strong performance on standard visual question answering benchmarks, yet their scalability under controlled structural complexity remains poorly understood.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.26029","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_trl-bench_b244de7b","familyId":"bmf_722ae64b1c7b","name":"TRL-Bench","oneLine":"TRL-Bench evaluates tabular encoders at the representation level by exporting row-, column-, or table-level embeddings through each encoder's supported wrapper and probing them with shared lightweight heads across three suites: TRL-CTbench (column/table), TRL-Rbench (row), and TRL-DLTE (compositional Data-Lake Table Enrichment) covering 16 tasks and 20 models.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09323","pdf":"https://arxiv.org/pdf/2606.09323","project":null,"code":"https://github.com/LOGO-CUHKSZ/TRL-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.09323"},"evidence":{"snippet":"We introduce TRL-Bench, a multi-granular tabular representation learning (TRL) benchmark that standardizes cross-paradigm representation-level evaluation: each encoder exports row-, column-, or table embeddings through its supported wrapper, and shared lightweight heads probe them across three suites: TRL-CTbench (column/table), TRL-Rbench (row), and TRL-DLTE (compositional Data-Lake Table Enrichment spanning all three granularities).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":53,"hfDailySubmittedAt":"2026-06-11T00:00:00.000Z","githubStars":10,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09323"},"ranking":{"90d":{"score":44,"rank":124,"coverage":0.7,"confidence":"Medium"}},"description":"TRL-Bench evaluates tabular encoders at the representation level by exporting row-, column-, or table-level embeddings through each encoder's supported wrapper and probing them with shared lightweight heads across three suites: TRL-CTbench (column/table), TRL-Rbench (row), and TRL-DLTE (compositional Data-Lake Table Enrichment) covering 16 tasks and 20 models.","whyItMatters":"Traditional end-to-end pipelines obscure the comparative quality of tabular encoders from different training paradigms. TRL-Bench provides a standardized protocol to isolate representation-level capability, enabling model selection based on task-specific strengths rather than a single aggregate score.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"75d17630044b0d81b27f75e8919bf9f8009ba719fd915f9545b8ae5a650f9be4"},"motivation":"Tabular encoders are usually evaluated inside task-specific end-to-end pipelines, so models from different training paradigms are difficult to compare directly even when they operate on similar tabular signals.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09323","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"LOGO Lab, CUHK-Shenzhen","organizationType":"academic-lab","sourceUrl":"https://github.com/LOGO-CUHKSZ/TRL-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_trustdabench_96f4ae98","familyId":"bmf_07f6231ee3bd","name":"TrustDABench","oneLine":"Evaluates LLM reliability and robustness for structured data analysis using 2,340 human-verified perturbed instances derived from evidence-path perturbations, scoring refusal behavior and resistance to table representation changes.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Robustness"],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-25","firstSeenAt":"2026-08-28","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.24145","pdf":"https://arxiv.org/pdf/2608.24145","project":null,"code":"https://github.com/Skyorca/TrustDABench","data":null,"hfPaper":"https://huggingface.co/papers/2608.24145"},"evidence":{"snippet":"We introduce TrustDABench, a benchmark that operationalizes these questions as reliability and robustness.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24145"},"ranking":{"30d":{"score":23,"rank":111,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":315,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates LLM reliability and robustness for structured data analysis using 2,340 human-verified perturbed instances derived from evidence-path perturbations, scoring refusal behavior and resistance to table representation changes.","whyItMatters":"Targets a practical gap in trustworthy data analysis where models may produce unsupported or inconsistent results across table forms.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:08:47.245811Z","inputHash":"6da3ea7b87151b64cc1079f5392ba72799d2e9769264f765b6df36b9bca77888"},"motivation":"LLMs are increasingly used to analyze spreadsheets, CSV files, and other structured data, but producing a correct-looking answer is not the same as producing a trustworthy analysis.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24145","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"attentionForecast":{"score":42,"confidence":"Medium","horizon":"7d","reason":"The structured data analysis reliability theme is broadly relevant, but the lack of confirmed public artifacts may limit early traction."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_76bf377f1a1ee3de","familyId":"catalog_family_76bf377f1a1ee3de","name":"TruthfulQA","oneLine":"TruthfulQA is a benchmark to measure whether language models are truthful in generating answers to questions. It comprises 817 questions that span 38 categories, including health, law, finance and politics. The questions are crafted such that some humans would answer falsely due to a false belief or misconception, testing models' ability to avoid generating false answers learned from human texts.","description":"TruthfulQA is a benchmark to measure whether language models are truthful in generating answers to questions. It comprises 817 questions that span 38 categories, including health, law, finance and politics. The questions are crafted such that some humans would answer falsely due to a false belief or misconception, testing models' ability to avoid generating false answers learned from human texts.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Knowledge","Legal","Reasoning","Finance","General","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2109.07958","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_76bf377f1a1ee3de"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/truthfulqa"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/truthfulqa"}],"catalogSources":[{"catalog":"benchlm","sourceId":"truthfulqa","url":"https://benchlm.ai/benchmarks/truthfulqa","paperUrl":"https://arxiv.org/abs/2109.07958","year":"2021","fullName":"TruthfulQA","format":"Question answering","tasks":"Truthfulness and misconception resistance","successorKey":null},{"catalog":"llm-stats","sourceId":"truthfulqa","url":"https://llm-stats.com/benchmarks/truthfulqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","legal","reasoning","finance","general","healthcare"],"catalogModelCount":18,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_ts-fault_c6d5ff11","familyId":"bmf_65f522aa76b1","name":"TS-Fault","oneLine":"Evaluates time series forecasting models under four explicit structural fault modes (time-warped shock, dependency-fracture shock, regime-transition missingness, cascading sensor-to-system failure) injected into lookback windows, with paired clean/corrupt protocol and five difficulty levels across nine datasets and six domains.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Robustness"],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18539","pdf":"https://arxiv.org/pdf/2606.18539","project":null,"code":"https://github.com/Ray-zyy/TS-Fault","data":null,"hfPaper":"https://huggingface.co/papers/2606.18539"},"evidence":{"snippet":"Treating TSF robustness as a data-quality problem, we present TS-Fault, a benchmark that evaluates forecasting models under explicit, parameterized fault scenarios with controllable semantic difficulty.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18539"},"ranking":{"90d":{"score":36,"rank":186,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates time series forecasting models under four explicit structural fault modes (time-warped shock, dependency-fracture shock, regime-transition missingness, cascading sensor-to-system failure) injected into lookback windows, with paired clean/corrupt protocol and five difficulty levels across nine datasets and six domains.","whyItMatters":"Standard clean-data leaderboards assume a single error metric predicts deployed reliability, but real faults are structured events. TS-Fault provides a diagnostic protocol that isolates robustness to named fault mechanisms at tunable severities, revealing that clean accuracy anti-correlates with robustness and mechanism-level faults reorder model rankings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"845f0288b326849a82f6bea76f8d141fc12f501023298f99bc795ce8cf2906ca"},"motivation":"Time series forecasting (TSF) underpins consequential decisions in energy, transportation, finance, and healthcare, yet TSF models are almost universally ranked by a single number (e.g., average error) on clean held-out data, under the implicit assumption that it predicts deployed reliability.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18539","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HKUST(GZ)","organizationType":"academic-lab","sourceUrl":"https://github.com/Ray-zyy/TS-Fault","role":"benchmark-publisher"}],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_ts-skill_e0ee25de","familyId":"bmf_214b8b173f4a","name":"TS-Skill","oneLine":"A controlled benchmark for evaluating analytical skills in time-series question answering: temporal scale selection, temporal localization, and cross-interval integration.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-05-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.24703","pdf":"https://arxiv.org/pdf/2605.24703","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.24703"},"evidence":{"snippet":"We introduce TS-Skill, a controlled benchmark for evaluating three composable analytical skills in TSQA: temporal scale selection (SK1), temporal localization (SK2), and cross-interval integration (SK3).","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.24703"},"ranking":{},"description":"A controlled benchmark for evaluating analytical skills in time-series question answering: temporal scale selection, temporal localization, and cross-interval integration.","whyItMatters":"Skill-level evaluation reveals temporal reasoning failures obscured by aggregate scores.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c1aea376bad770f9fb95d59c151329c5270e00ac8a067e369194e3681e82b445"},"motivation":"Large language models (LLMs) and time-series language models (TSLMs) are increasingly applied to time-series question answering (TSQA).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.24703","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_tsai-metafraud_9ef6a40f","familyId":"bmf_f0cefb4e8b91","name":"TSAI-MetaFraud","oneLine":"TSAI-MetaFraud is a multimodal, multi-task benchmark dataset for fraud detection in virtual economies, integrating behavioral, transactional, and graph-structured data. It defines tasks including fraud detection, node classification, temporal link prediction, and weakly supervised fraud detection, with baseline evaluations.","area":"Multimodal","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":["Financial Services"],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.09528","pdf":"https://arxiv.org/pdf/2607.09528","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.09528"},"evidence":{"snippet":"To address this gap, we present TSAI-MetaFraud, a multimodal, multi-task benchmark dataset for fraud analytics in virtual economies.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.09528"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TSAI-MetaFraud is a multimodal, multi-task benchmark dataset for fraud detection in virtual economies, integrating behavioral, transactional, and graph-structured data. It defines tasks including fraud detection, node classification, temporal link prediction, and weakly supervised fraud detection, with baseline evaluations.","whyItMatters":"Fraud in metaverse ecosystems is a new challenge that combines behavioral and financial data. TSAI-MetaFraud provides a unified benchmark to advance multimodal learning and fraud analytics in virtual economies, filling a gap in existing datasets.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f74e2f7e191243c47e047ab633312cd461986a90916a99261771bff911325246"},"motivation":"The emergence of metaverse platforms has created virtual economies that introduce new challenges related to fraud, bot activity, and illicit financial behavior.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.09528","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_tsc-bench_7fff3bb6","familyId":"bmf_21744c0a2f0c","name":"TSC-Bench","oneLine":"TSC-Bench is a benchmark for triple-shot composition, generating establishing, medium, and close-up crops from human-centric images with shot descriptions. It contains 1.2k expert-annotated test cases.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05635","pdf":"https://arxiv.org/pdf/2606.05635","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05635"},"evidence":{"snippet":"In addition, we present TSC-Bench, a benchmark of 1.2k expert-annotated test cases.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05635"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TSC-Bench is a benchmark for triple-shot composition, generating establishing, medium, and close-up crops from human-centric images with shot descriptions. It contains 1.2k expert-annotated test cases.","whyItMatters":"Multi-shot composition is valuable for creative workflows, but existing benchmarks focus on single crops. TSC-Bench could evaluate narrative-driven cropping capabilities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"25d88d833e17252c7be1a20a6e93e908384c8512374c7efd4c8d6ede5e482c2b"},"motivation":"Prior work on aesthetic composition typically produces a single aesthetically pleasing crop, overlooking the narrative value of composing multiple shots from one scene.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05635","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_tsm-bench_b1598d50","familyId":"bmf_dd5ec332da38","name":"TSM-Bench","oneLine":"TSM-Bench is a multilingual, multi-generator, multi-task benchmark for evaluating machine-generated text detectors on real-world Wikipedia editing tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.31113","pdf":"https://arxiv.org/pdf/2605.31113","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.31113"},"evidence":{"snippet":"We introduce \\textsc{TSM-Bench}, a multilingual, multi-generator, and \\textit{multi-task} benchmark for evaluating MGT detectors on common, real-world Wikipedia editing tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.31113"},"ranking":{},"description":"TSM-Bench is a multilingual, multi-generator, multi-task benchmark for evaluating machine-generated text detectors on real-world Wikipedia editing tasks.","whyItMatters":"Reveals performance drops in detectors on task-specific MGT, providing a foundation for developing robust detectors for UGC platforms.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c53c900004dbc2aaa4c8a856779bce8f2e452dddc6c07d96e7e33b2f83c6613b"},"motivation":"Automatically detecting machine-generated text (MGT) is critical to maintaining the knowledge integrity of user-generated content (UGC) platforms such as Wikipedia.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.31113","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_tsugo_41aa5b88","familyId":"bmf_9964a7463fc4","name":"TsuGO","oneLine":"TsuGO is a process-level reasoning benchmark for search efficiency in LLMs using Go life-and-death problems, parsing CoT into search trees and reporting diagnostics.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.13221","pdf":"https://arxiv.org/pdf/2608.13221","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.13221"},"evidence":{"snippet":"We introduce TsuGO, a process-level reasoning benchmark for evaluating Search Efficiency in LLM reasoning through Go life-and-death problems.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13221"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TsuGO is a process-level reasoning benchmark for search efficiency in LLMs using Go life-and-death problems, parsing CoT into search trees and reporting diagnostics.","whyItMatters":"It adds search organization as a missing evaluation dimension beyond final-answer accuracy in LLM reasoning benchmarks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"65fed3fb15917c25e4c2074bc3b207b246c4207ce6f8d6babebde1b9fd1b6780"},"motivation":"The evaluation of LLM reasoning is moving from final-answer accuracy to process-level assessment, yet existing methods still fail to capture how models plan reasoning paths and allocate reasoning resources--that is, how they organize search.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13221","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_tswap-a-multilingual-retrieval-augmented-t_149c0a74","familyId":"bmf_5ce2d10fda86","name":"TSWAP Thai Wellness Benchmark","oneLine":"TSWAP Thai Wellness Benchmark is a retrieval-augmented question-answering benchmark for Thai traditional medicine and wellness advice. It includes 50 questions with gold document IDs, evaluated via Recall@5, and also releases production QA logs and a no-retrieval probe.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Information retrieval","Factuality"],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":0.55,"links":{"report":"http://arxiv.org/abs/2608.22917v1","pdf":"https://arxiv.org/pdf/2608.22917v1","project":null,"code":null,"data":"https://huggingface.co/datasets/iapp/tswap-wellness-benchmark","hfPaper":null},"evidence":{"snippet":"We release the first Thai traditional-medicine/wellness retrieval benchmark (50 questions with gold document IDs; Recall@5 = 0.88), production QA logs (91.1% test-retest pass over 259 cases), and a 71-question frontier no-retrieval probe showing what each grounding pillar contributes: without the safety prompt the backend model family produced a full drug-dosing schedule and complied with out-of-scope requests, and without the knowledge base it produced zero verifiable provider recommendations.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":0,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2608.22917"},"ranking":{"today":{"score":46,"rank":12,"coverage":0.45,"confidence":"Medium","datasetDownloadRank":1,"datasetRankPopulation":1},"30d":{"score":43,"rank":93,"coverage":0.15,"confidence":"Low","datasetDownloadRank":21,"datasetRankPopulation":21},"90d":{"score":35,"rank":299,"coverage":0.3,"confidence":"Low","datasetDownloadRank":51,"datasetRankPopulation":51}},"description":"TSWAP Thai Wellness Benchmark is a retrieval-augmented question-answering benchmark for Thai traditional medicine and wellness advice. It includes 50 questions with gold document IDs, evaluated via Recall@5, and also releases production QA logs and a no-retrieval probe.","whyItMatters":"The benchmark fills a gap in multilingual wellness and safety evaluation, providing a public dataset and standardized protocol to assess retrieval grounding and safety in LLM-based systems. It supports research in low-resource language NLP and responsible AI deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"e7ea1740d84356477db68a757ceb7674c07a57214b80b294364d8951c2b635d3"},"motivation":"We present TSWAP, a deployed eight-language conversational wellness advisor grounded, via retrieval-augmented generation, in a verified knowledge base of Thai traditional medicine and certified wellness providers.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Released dataset on Hugging Face with clear metrics and a formal README name.","canonicalNameSource":"official_readme","canonicalNameEvidence":"pretty_name: TSWAP Thai Wellness Benchmark & Evaluation Logs"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22917v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":40,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets a specific niche in Thai wellness and retrieval, with public data release but limited general appeal."},"evaluationMode":"public_reusable","publishers":[{"name":"IAPP","organizationType":"company-research-lab","sourceUrl":"https://huggingface.co/datasets/iapp/tswap-wellness-benchmark","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Safety & Trustworthiness","Search & Retrieval"],"domainScope":"specific"},{"id":"bm_tua-bench_56e7e977","familyId":"bmf_6458a26e1270","name":"TUA-Bench","oneLine":"Evaluates terminal-use agents on 120 real-world tasks across five families (document editing, email, web info, scientific/engineering workflows) using execution-based scoring.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.28480","pdf":"https://arxiv.org/pdf/2606.28480","project":"https://www.tuabench.ai","code":"https://github.com/facebookresearch/TUA-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2606.28480"},"evidence":{"snippet":"We introduce TUA-Bench, a general-purpose benchmark for terminal-use agents.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":48,"hfDailySubmittedAt":"2026-06-30T00:00:00.000Z","githubStars":47,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.28480"},"ranking":{"90d":{"score":54,"rank":39,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates terminal-use agents on 120 real-world tasks across five families (document editing, email, web info, scientific/engineering workflows) using execution-based scoring.","whyItMatters":"Provides a broad, realistic terminal benchmark beyond coding, showing frontier agents achieve only 65.8% success, highlighting gaps in general-purpose digital work.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6b4c90791bb32c4e309ff9509a785f3211b5bd10d48a36c6d542ace7f7d1455d"},"motivation":"As large language models and harness frameworks continue to advance, agents operating in terminals are increasingly capable of performing a broader range of general computer-use tasks beyond coding.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.28480","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Facebook Research","organizationType":"company-research-lab","sourceUrl":"https://github.com/facebookresearch/TUA-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_tukabench_beb4bae4","familyId":"bmf_65c1dc6edc86","name":"TukaBench","oneLine":"TUKABENCH is a benchmark for jailbreak evaluation in seven African languages, extending JailbreakBench with human translations, cultural adaptations, and code-switched prompts. It assesses LLM safety in low-resource languages using metrics like Refused, Jailbroken, and Deflection, with human validation of LLM-as-a-judge.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-31","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01322","pdf":"https://arxiv.org/pdf/2606.01322","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01322"},"evidence":{"snippet":"We introduce TUKABENCH, a jailbreak benchmark for seven African languages that extends JailbreakBench (JBB) beyond direct translation through four settings: human translation of JBB prompts, English adaptation to African contexts followed by human translation, human-curated prompts validated through interactions with GPT-5.2, and code-switched prompts combining English and African languages, isolating the effect of language, cultural grounding, and prompt evasiveness on model safety.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01322"},"ranking":{},"description":"TUKABENCH is a benchmark for jailbreak evaluation in seven African languages, extending JailbreakBench with human translations, cultural adaptations, and code-switched prompts. It assesses LLM safety in low-resource languages using metrics like Refused, Jailbroken, and Deflection, with human validation of LLM-as-a-judge.","whyItMatters":"The benchmark addresses a gap in safety evaluation for low-resource African languages, showing that models are more vulnerable to jailbreak prompts in these languages. It could inform safer deployment of LLMs in multilingual contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"10d2859030b484f55598584f77c60bfaeaac1940867c15577f58e92d3b7d502c"},"motivation":"Safety evaluation of Large Language Models (LLMs) remains heavily English-centric, leaving Low-Resource Languages (LRLs), particularly African ones, critically underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01322","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_turkish-llm-benchmark_fa8945d1","familyId":"bmf_ddbc317711bc","name":"Turkish LLM Benchmark","oneLine":"Evaluates Turkish language quality, instruction following, latency, API cost, local throughput, and speculative decoding across OpenAI-compatible LLM endpoints.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/kadirnar/turkish-llm-benchmark","pdf":null,"project":"https://docs.astral.sh/uv/","code":"https://github.com/kadirnar/turkish-llm-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"turkish-llm-benchmark Provider-neutral Turkish quality, cost, latency, and performance benchmarks for LLMs benchmark evaluation llm openrouter turkish # Turkish LLM Benchmark [![CI](https://github.com/kadirnar/turkish-llm-benchmark/actions/workflows/ci.yml/badge.svg)](https://github.com/kadirnar/turkish-llm-benchmark/actions/workflows/ci.yml) A provider-neutral, reproducible benchmark suite for measuring Turkish language quality, instruction following, latency, API cost, local throughput, and sp","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:kadirnar/turkish-llm-benchmark"},"ranking":{"30d":{"score":28,"rank":87,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":257,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates Turkish language quality, instruction following, latency, API cost, local throughput, and speculative decoding across OpenAI-compatible LLM endpoints.","whyItMatters":"It provides a provider-neutral suite for measuring Turkish capabilities and costs, which are underrepresented in general benchmark suites.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"b172c58a769ee4f2f056cc263747ec671928c41ac2448d829e8565b700822941"},"motivation":"turkish-llm-benchmark Provider-neutral Turkish quality, cost, latency, and performance benchmarks for LLMs benchmark evaluation llm openrouter turkish # Turkish LLM Benchmark [![CI](https://github.com/kadirnar/turkish-llm-benchmark/actions/workflows/ci.yml/badge.svg)](https://github.com/kadirnar/turkish-llm-benchmark/actions/workflows/ci.yml) A provider-neutral, reproducible benchmark suite for measuring Turkish lan…","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/kadirnar/turkish-llm-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":48,"confidence":"Medium","horizon":"7d","reason":"A well-documented multilingual benchmark for Turkish fills a niche, though its audience may be regional and moderately sized."},"evaluationMode":"score_submission","publishers":[{"name":"kadirnar","organizationType":"community","sourceUrl":"https://github.com/kadirnar/turkish-llm-benchmark","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_turnbench_084201bf","familyId":"bmf_2b94493a4968","name":"TurnBench","oneLine":"Evaluates end-of-turn and interruption detection in dyadic human conversation across six interaction styles using a 30-hour hand-labeled corpus and standardized metrics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["eess.AS"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-27","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25218","pdf":"https://arxiv.org/pdf/2608.25218","project":"https://turnbench.sesame.com","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To address this, we present TurnBench, a multi-domain benchmark that pairs a 30-hour, hand-labeled corpus of dyadic human conversation with a standardized evaluation protocol for end-of-turn and interruption detection.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25218"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates end-of-turn and interruption detection in dyadic human conversation across six interaction styles using a 30-hour hand-labeled corpus and standardized metrics.","whyItMatters":"Provides the first multi-domain, triple-annotated corpus for turn-taking, enabling consistent comparison across conversation types and revealing type-dependent interruption errors.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"2b00ba9b372106bee99f6a1046797422aebe846738d80f44af424a86e9e81348"},"motivation":"Speakers in natural conversation take turns speaking and listening, deciding in real time when to take, hold, or yield the floor.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"Includes a released corpus, training set, public leaderboard, and standardized evaluation protocol; meets repeatable benchmarking and public comparison criteria.","canonicalNameSource":"paper_title","canonicalNameEvidence":"TurnBench: A Multi-Domain Benchmark for Turn-Taking Dynamics in Spoken Dialogue"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25218","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"Spoken dialogue benchmarking is a growing area with a clear leaderboard and multi-domain scope, likely to attract moderate early interest from speech and NLP communities."},"evaluationMode":"score_submission","publishers":[{"name":"Sesame","organizationType":"company-research-lab","sourceUrl":"https://turnbench.sesame.com","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_turtleai_8dfa596d","familyId":"bmf_02b159b6d40e","name":"TurtleAI","oneLine":"TurtleAI is a benchmark of 823 tasks for visual programming in Turtle Graphics, evaluating models on perceiving geometric patterns and synthesizing Python code.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Code generation"],"topics":["Multimodal"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03626","pdf":"https://arxiv.org/pdf/2606.03626","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03626"},"evidence":{"snippet":"To bridge this gap, we introduce TurtleAI, a benchmark containing 823 tasks curated based on real-world visual programming tasks in the Turtle Graphics domain.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03626"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"TurtleAI is a benchmark of 823 tasks for visual programming in Turtle Graphics, evaluating models on perceiving geometric patterns and synthesizing Python code.","whyItMatters":"Bridges the gap in education-oriented visual programming evaluation, highlighting limitations in spatial reasoning and code generation for VLMs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f8462c60a063d61d884c72dfdfeb51f64e31bcffd539d60cb77ccab35cdc6a49"},"motivation":"Vision-language models (VLMs) have been explored for visual programming, where they generate code to solve visual tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03626","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_221f38ca3f33898d","familyId":"catalog_family_221f38ca3f33898d","name":"TVBench","oneLine":"TVBench is a temporal video understanding benchmark evaluating reasoning over actions, events, and temporal dynamics in videos.","description":"TVBench is a temporal video understanding benchmark evaluating reasoning over actions, events, and temporal dynamics in videos.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tvbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_221f38ca3f33898d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tvbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tvbench","url":"https://llm-stats.com/benchmarks/tvbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_tw-bench_62d8b3a2","familyId":"bmf_be1e92b1a963","name":"tw-bench","oneLine":"Taiwan benchmark dataset with claimed leaderboard and methodology on an external site.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://huggingface.co/datasets/OpenTWBench/tw-bench","pdf":null,"project":null,"code":null,"data":"https://huggingface.co/datasets/OpenTWBench/tw-bench","hfPaper":null},"evidence":{"snippet":"🔗 opentwbench.ai · leaderboard, methodology, submissions from datasets import load_dataset ds = load_dataset(\"OpenTWBench/tw-bench\", \"dentistry\", split=\"test\") ds = load_dataset(\"OpenTWBench/tw-bench\", \"field_medicine\", split=\"test\") ds =… See the full description on the dataset page: https://huggingface.co/datasets/OpenTWBench/tw-bench.","reasonCodes":["discovered via huggingface","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":62,"hfDatasetLikes":0},"source":{"type":"huggingface","id":"huggingface:opentwbench/tw-bench"},"ranking":{"30d":{"score":49,"rank":35,"coverage":0.15,"confidence":"Low","datasetDownloadRank":27,"datasetRankPopulation":30},"90d":{"score":45,"rank":108,"coverage":0.3,"confidence":"Low","datasetDownloadRank":56,"datasetRankPopulation":66}},"description":"Taiwan benchmark dataset with claimed leaderboard and methodology on an external site.","whyItMatters":"Potential localized evaluation resource, but unverified dataset alone does not establish a formal benchmark release.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"64f6d3d481891a9319326729298ddba2126759bd558600533cb93247918f489b"},"motivation":"OpenTWBench 255,854 multiple-choice questions across 148 academic subjects, measuring what a language model knows about Taiwan — in Traditional Chinese, from the national examinations that license Taiwan's regulated professions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"Only an unverified Hugging Face dataset link is present; no independent source confirms the benchmark's name or formal release."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://huggingface.co/datasets/OpenTWBench/tw-bench","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":0,"confidence":"Low","horizon":"7d","reason":"Excluded due to deferred status and insufficient evidence to forecast attention."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_tw-exam-bench_0b096fe2","familyId":"bmf_2fc7081321fe","name":"tw-exam-bench","oneLine":"Taiwan exam benchmark dataset with claimed leaderboard and methodology on an external site.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-24","recognitionConfidence":0.4,"links":{"report":"https://huggingface.co/datasets/OpenTWBench/tw-exam-bench","pdf":null,"project":null,"code":null,"data":"https://huggingface.co/datasets/OpenTWBench/tw-exam-bench","hfPaper":null},"evidence":{"snippet":"🔗 opentwbench.ai · leaderboard, methodology, submissions from datasets import load_dataset ds = load_dataset(\"OpenTWBench/tw-exam-bench\", \"dentistry\", split=\"test\") ds = load_dataset(\"OpenTWBench/tw-exam-bench\", \"field_medicine\", split=\"test\") ds =… See the full description on the dataset page: https://huggingface.co/datasets/OpenTWBench/tw-exam-bench.","reasonCodes":["discovered via huggingface","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":62,"hfDatasetLikes":0},"source":{"type":"huggingface","id":"huggingface:opentwbench/tw-exam-bench"},"ranking":{"30d":{"score":49,"rank":34,"coverage":0.15,"confidence":"Low","datasetDownloadRank":26,"datasetRankPopulation":30},"90d":{"score":45,"rank":107,"coverage":0.3,"confidence":"Low","datasetDownloadRank":55,"datasetRankPopulation":66}},"description":"Taiwan exam benchmark dataset with claimed leaderboard and methodology on an external site.","whyItMatters":"Could provide localized exam-style evaluation, but insufficient details are available to confirm scoring contract or benchmark stability.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-26T06:15:55.708965Z","inputHash":"8cc68e7d94cbb02ab70c0494cb6a851457028510be56145bf83a97e1454a2746"},"motivation":"OpenTWBench 255,854 multiple-choice questions across 148 academic subjects, measuring what a language model knows about Taiwan — in Traditional Chinese, from the national examinations that license Taiwan's regulated professions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-26T06:15:55.708965Z","model":"deepseek-v4-pro","decisionReason":"Only an unverified Hugging Face dataset link is present; no independent source confirms the benchmark's name or formal release."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://huggingface.co/datasets/OpenTWBench/tw-exam-bench","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-24T07:04:32.475961Z"},"attentionForecast":{"score":0,"confidence":"Low","horizon":"7d","reason":"Excluded due to deferred status and insufficient evidence to forecast attention."},"evaluationMode":"public_reusable","displayEligible":false,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_b8088edd09d095bd","familyId":"catalog_family_b8088edd09d095bd","name":"TydiQA","oneLine":"A multilingual question answering benchmark covering 11 typologically diverse languages with 204K question-answer pairs. Questions are written by people seeking genuine information and data is collected directly in each language without translation to test model generalization across diverse linguistic structures.","description":"A multilingual question answering benchmark covering 11 typologically diverse languages with 204K question-answer pairs. Questions are written by people seeking genuine information and data is collected directly in each language without translation to test model generalization across diverse linguistic structures.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/tydiqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b8088edd09d095bd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/tydiqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"tydiqa","url":"https://llm-stats.com/benchmarks/tydiqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","reasoning"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ua-legal-bench_a6ec4166","familyId":"bmf_2308fd0ed61a","name":"UA-Legal-Bench","oneLine":"UA-Legal-Bench is a five-task benchmark for evaluating LLMs on Ukrainian legal reasoning using court decisions, covering classification, outcome prediction, and norm extraction.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.29170","pdf":"https://arxiv.org/pdf/2605.29170","project":null,"code":null,"data":"https://huggingface.co/datasets/overthelex/ua-legal-bench","hfPaper":"https://huggingface.co/papers/2605.29170"},"evidence":{"snippet":"We introduce UA-Legal-Bench, a five-task benchmark for evaluating large language models on Ukrainian legal reasoning, built from the Unified State Register of Court Decisions (EDRSR) -- one of the world's largest open judicial corpora (99.5 million decisions).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":125,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2605.29170"},"ranking":{},"description":"UA-Legal-Bench is a five-task benchmark for evaluating LLMs on Ukrainian legal reasoning using court decisions, covering classification, outcome prediction, and norm extraction.","whyItMatters":"Addresses the dearth of legal NLP benchmarks for non-English, morphologically rich languages, and provides insights into few-shot effects and model scaling in legal tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e978ff6afdfdcdf1dc47a20a0ed30c25f1ac1b7881c123b03f722f73e8691bd4"},"motivation":"Legal NLP benchmarks are overwhelmingly English-centric, leaving failure modes in morphologically rich, non-Latin-script languages undetected.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.29170","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"overthelex","organizationType":"community","sourceUrl":"https://huggingface.co/datasets/overthelex/ua-legal-bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_uav-ovo_af2dbcd6","familyId":"bmf_90a06c92b3f9","name":"UAV-OVO","oneLine":"UAV-OVO is an out-of-viewpoint generalization benchmark for UAV action recognition. It evaluates models on low-depression viewpoint training data and tests on high-depression viewpoint out-of-distribution data, with class-matched ID/OOD splits.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.25615","pdf":"https://arxiv.org/pdf/2605.25615","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.25615"},"evidence":{"snippet":"We introduce UAV-OVO, an Out-of-Viewpoint generalization benchmark for UAV action recognition.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25615"},"ranking":{},"description":"UAV-OVO is an out-of-viewpoint generalization benchmark for UAV action recognition. It evaluates models on low-depression viewpoint training data and tests on high-depression viewpoint out-of-distribution data, with class-matched ID/OOD splits.","whyItMatters":"Standard UAV action recognition benchmarks often overlook viewpoint shifts, leading to models that rely on viewpoint-specific shortcuts. UAV-OVO provides a controlled testbed to measure and improve robustness to viewpoint changes, which is critical for real-world UAV deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d17b74d39b18405218847cb9482d903f3d40c346ac5d1150fd7886bcc3c7dece"},"motivation":"UAV action recognition faces a deployment shift that standard benchmarks often obscure: a model trained on UAV footage captured from low-depression viewpoints may be required to recognize the same action classes from high-depression viewpoints.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25615","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_uav3dcrop_4b4a17ef","familyId":"bmf_fbd68110137b","name":"UAV3DCrop","oneLine":"UAV3DCrop is a benchmark of repeated multi-angle UAV crop surveys with 88,830 images across 91 scenes, evaluating 3D reconstruction methods on appearance, geometry, and canopy height.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.06404","pdf":"https://arxiv.org/pdf/2608.06404","project":"https://link-dev.github.io/UAV3DCrop/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.06404"},"evidence":{"snippet":"We introduce UAV3DCrop, a public benchmark of repeated multi-angle unmanned aerial vehicle (UAV) crop surveys.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.06404"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"UAV3DCrop is a benchmark of repeated multi-angle UAV crop surveys with 88,830 images across 91 scenes, evaluating 3D reconstruction methods on appearance, geometry, and canopy height.","whyItMatters":"Provides a domain-specific benchmark for agronomically relevant 3D reconstruction, revealing that generic methods do not directly translate to crop monitoring.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9658c34d4f7bdd6619c069d200edba2f4f07787bf6607b4e2a0a30e0eb76b7db"},"motivation":"Accurate 3D crop monitoring underpins data-driven precision agriculture by enabling field-scale analysis of plant structure, growth dynamics, and management response.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.06404","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_uesf-bench_80fcb44c","familyId":"bmf_d538adf3f888","name":"UESF-Bench","oneLine":"UESF-Bench evaluates embodied agents on unified language-guided human seeking and following in dynamic environments, covering semantic-guided exploration, behavior switching, and identity grounding across single- and multi-person settings. Scoring uses success metrics for both phases.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.13621","pdf":"https://arxiv.org/pdf/2607.13621","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.13621"},"evidence":{"snippet":"To address these limitations, we introduce the Unified Embodied Seeking and Following Benchmark (UESF-Bench), a large-scale and diverse benchmark for embodied human seeking and following.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.13621"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"UESF-Bench evaluates embodied agents on unified language-guided human seeking and following in dynamic environments, covering semantic-guided exploration, behavior switching, and identity grounding across single- and multi-person settings. Scoring uses success metrics for both phases.","whyItMatters":"Existing benchmarks assume the target is visible at start, missing realistic scenarios where agents must first find and then follow. UESF-Bench provides a unified evaluation to advance embodied agents in more practical human-robot interaction.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"72f1b5c64cec3d5a46ad1f6d58b752fbf886ec8273d3a3fde926a5b9df17e7d2"},"motivation":"Language-guided human following is an important capability for embodied agents, but existing benchmarks typically assume that the target person is visible at the start of an episode.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.13621","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_uhi-bench_9c52740b","familyId":"bmf_854f936aa343","name":"UHI-Bench","oneLine":"UHI-Bench is a benchmark for dual-source urban heat island (UHI) modeling that integrates dynamic meteorological drivers and static urban morphology. It evaluates over 20 baselines from four model families across five tasks on 20 cities in nine Köppen climate classes, covering land surface temperature UHI and near-surface air temperature UHI.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.23857","pdf":"https://arxiv.org/pdf/2608.23857","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To bridge these gaps, we introduce UHI-Bench, the first UHI benchmark for dual-source UHI modeling that integrates dynamic and static environmental context.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23857"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"UHI-Bench is a benchmark for dual-source urban heat island (UHI) modeling that integrates dynamic meteorological drivers and static urban morphology. It evaluates over 20 baselines from four model families across five tasks on 20 cities in nine Köppen climate classes, covering land surface temperature UHI and near-surface air temperature UHI.","whyItMatters":"The benchmark addresses the lack of standardized evaluation for dual-source UHI modeling, which is crucial for understanding urban heat exposure and climate adaptation. It provides a unified framework to facilitate cross-city transfer and climate data equity, guiding practical urban heat mitigation strategies.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"fdf46721c81ed16bf6663d46bc344d284408c8015d8beca31e072f47901833b6"},"motivation":"Urban heat islands (UHIs) are intensifying under climate change, exacerbating thermal exposure risks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with standardized pipeline, multiple baselines, and public dataset release.","canonicalNameSource":"abstract","canonicalNameEvidence":"we introduce UHI-Bench, the first UHI benchmark for dual-source UHI modeling"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.23857","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":45,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses climate-related challenges with broad application potential, but may attract niche specialist attention initially."},"evaluationMode":"score_submission","publishers":[{"name":"UHI-Bench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.23857","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_ui2app_cf415439","familyId":"bmf_aca78e2fdb0c","name":"UI2App","oneLine":"Benchmarks visual interaction inference in executable web application generation. Contains 327 screenshots in 45 sets, evaluating executability, navigation reachability, visual fidelity, and interaction inference via the IIS metric.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.06306","pdf":"https://arxiv.org/pdf/2607.06306","project":null,"code":"https://github.com/chenmancm169/UI2App","data":null,"hfPaper":"https://huggingface.co/papers/2607.06306"},"evidence":{"snippet":"To address this gap, we introduce UI2App, the first benchmark targeting interaction inference, the ability to recover application behavior from screenshots alone, without any textual or behavioral guidance.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":"2026-07-21T00:00:00.000Z","githubStars":3,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06306"},"ranking":{"90d":{"score":29,"rank":249,"coverage":0.7,"confidence":"Medium"}},"description":"Benchmarks visual interaction inference in executable web application generation. Contains 327 screenshots in 45 sets, evaluating executability, navigation reachability, visual fidelity, and interaction inference via the IIS metric.","whyItMatters":"Targets a gap in measuring behavior inference from screenshots, not just visual fidelity. Could help assess models' ability to produce interactive applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1ddac74dcec827323df25757304b635430681d503488d8ea5f382cda06cfa635"},"motivation":"Large language models (LLMs) have demonstrated growing competence in web page generation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06306","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_ultravr_947de071","familyId":"bmf_cea87e46d51d","name":"UltraVR","oneLine":"UltraVR is a diagnostic benchmark for evidence-grounded visual reasoning over ultra-resolution images, spanning four domains: CCTV surveillance, remote sensing, whole-slide pathology, and industrial anomaly detection. It includes structured ground-truth chain-of-thought annotations for process-level diagnosis.","area":"Vision & 3D","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05576","pdf":"https://arxiv.org/pdf/2606.05576","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05576"},"evidence":{"snippet":"We introduce UltraVR, a diagnostic benchmark for evidence-grounded visual reasoning over ultra-resolution images.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05576"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"UltraVR is a diagnostic benchmark for evidence-grounded visual reasoning over ultra-resolution images, spanning four domains: CCTV surveillance, remote sensing, whole-slide pathology, and industrial anomaly detection. It includes structured ground-truth chain-of-thought annotations for process-level diagnosis.","whyItMatters":"Standard VQA benchmarks report only final accuracy, obscuring whether models acquire and integrate visual evidence. UltraVR's process-level annotations could help localize failures in ultra-resolution reasoning pipelines.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"26a007abbc72b1b5ef7cfb688409853a0dd211c8af425f162a6f8c72365a4472"},"motivation":"Vision-language models (VLMs) excel on visual question answering and multimodal reasoning benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05576","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_umi-bench_1bf436a3","familyId":"bmf_b7b6ea4bab62","name":"UMI-Bench","oneLine":"UMI-Bench 1.0 is a benchmark for tabletop robotic manipulation policies using the Universal Manipulation Interface, with a unified protocol for data collection, reset, execution, and logging.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.10382","pdf":"https://arxiv.org/pdf/2606.10382","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.10382"},"evidence":{"snippet":"We present UMI-Bench 1.0, a local-first real-robot benchmark for standardized evaluation of UMI-style manipulation policies.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.10382"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"UMI-Bench 1.0 is a benchmark for tabletop robotic manipulation policies using the Universal Manipulation Interface, with a unified protocol for data collection, reset, execution, and logging.","whyItMatters":"Real-robot evaluation is crucial for assessing manipulation policies beyond curated demos; UMI-Bench aims to standardize evaluation for UMI-style policies, which could aid comparisons and reproducibility.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6079c9110175d6d6a2fb2d02525386190a09fdfd6523782622625793def40019"},"motivation":"Real-robot evaluation is essential for understanding whether learned manipulation policies can operate reliably outside curated demonstrations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.10382","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_underspecbench_dd8281f2","familyId":"bmf_81aeb66da19a","name":"UnderSpecBench","oneLine":"UnderSpecBench evaluates coding agents on DevOps tasks under varying instruction underspecification, measuring action-boundary violations such as wrong-target or over-scope actions.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.02294","pdf":"https://arxiv.org/pdf/2607.02294","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.02294"},"evidence":{"snippet":"We present UnderSpecBench, a benchmark for measuring action-boundary violations in coding agents (i.e., Claude Code, Codex, and OpenCode) on DevOps tasks.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.02294"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"UnderSpecBench evaluates coding agents on DevOps tasks under varying instruction underspecification, measuring action-boundary violations such as wrong-target or over-scope actions.","whyItMatters":"Existing agent benchmarks focus on task completion, potentially overstating safe autonomy. UnderSpecBench highlights the gap in measuring safe behavior under underspecified instructions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7a3e3adb24badfc1e0b505199681d83c214c2cf7d2e163d3a14c224eb601b117"},"motivation":"LLM coding agents are increasingly deployed to act autonomously on real production infrastructure.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.02294","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_unicad_f706b085","familyId":"bmf_b2109c024c36","name":"UniCAD","oneLine":"UniCAD is a benchmark for multi-modal CAD learning covering point-to-CAD reconstruction, text/image-to-CAD generation, and CAD question answering. It includes a universal model, UniCAD-MLLM, and reports state-of-the-art results on UniCAD and Fusion360 benchmarks.","area":"Science & Engineering","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Manufacturing"],"capabilities":["Geometric reasoning"],"topics":["CAD"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05058","pdf":"https://arxiv.org/pdf/2606.05058","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05058"},"evidence":{"snippet":"To address this gap, we introduce UniCAD, a comprehensive benchmark for multi-modal CAD learning that covers point-to-CAD reconstruction, text/image-to-CAD generation, and CAD question answering across diverse input modalities.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05058"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"UniCAD is a benchmark for multi-modal CAD learning covering point-to-CAD reconstruction, text/image-to-CAD generation, and CAD question answering. It includes a universal model, UniCAD-MLLM, and reports state-of-the-art results on UniCAD and Fusion360 benchmarks.","whyItMatters":"CAD research lacks a unified benchmark for multi-modal, multi-task learning. UniCAD aims to fill this gap by providing a comprehensive evaluation suite across diverse tasks and modalities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e8ae7a82b3fc7e6a12df8a3a36fa623bdfd36faa44407c0a0dbe26190ac36fb2"},"motivation":"Computer-Aided Design (CAD) underpins modern engineering and manufacturing by enabling the creation of precise, editable 3D models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.05058","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_uniclawbench_4d04906d","familyId":"bmf_36cef9ecb058","name":"UniClawBench","oneLine":"A capability-driven benchmark for proactive agents in real-world tasks, with 400 bilingual tasks across five capabilities. It evaluates agents in Docker containers using step-by-step checkpoints and a closed-loop strategy with executor, supervisor, and user agents.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.08768","pdf":"https://arxiv.org/pdf/2607.08768","project":"https://uniclawbench.github.io","code":"https://github.com/HKU-MMLab/UniClawBench","data":null,"hfPaper":"https://huggingface.co/papers/2607.08768"},"evidence":{"snippet":"To address these limitations, we introduce UniClawBench, the first capability-driven benchmark designed to evaluate proactive agents in dynamic, real-world settings.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":34,"hfDailySubmittedAt":null,"githubStars":39,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.08768"},"ranking":{"90d":{"score":52,"rank":53,"coverage":0.7,"confidence":"Medium"}},"description":"A capability-driven benchmark for proactive agents in real-world tasks, with 400 bilingual tasks across five capabilities. It evaluates agents in Docker containers using step-by-step checkpoints and a closed-loop strategy with executor, supervisor, and user agents.","whyItMatters":"Existing agent benchmarks rely on sandboxed environments and single-turn paradigms, which do not reflect real-world complexity. This benchmark provides a dynamic, capability-based evaluation that helps compare models and agent frameworks, aiding in identifying failure root causes.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2f779a2c4ca85aa66b2e5e837c8b79ccc20b727c7fcd6b6b7d04b6ff494ace17"},"motivation":"The rapid development of large language models and multimodal large language models has accelerated the emergence of proactive agents capable of operating everyday tools and assisting users in real-world environments.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.08768","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"HKU-MMLab","organizationType":"academic-lab","sourceUrl":"https://github.com/HKU-MMLab/UniClawBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_5751aeb45604a440","familyId":"catalog_family_5751aeb45604a440","name":"Uniform Bar Exam","oneLine":"The Uniform Bar Examination (UBE) benchmark evaluates language models on the complete bar exam including multiple-choice Multistate Bar Examination (MBE), open-ended Multistate Essay Exam (MEE), and Multistate Performance Test (MPT) components. Used to assess legal reasoning capabilities across seven subject areas including Evidence, Torts, Constitutional Law, Contracts, Criminal Law and Procedure, Real Property, and Civil Procedure.","description":"The Uniform Bar Examination (UBE) benchmark evaluates language models on the complete bar exam including multiple-choice Multistate Bar Examination (MBE), open-ended Multistate Essay Exam (MEE), and Multistate Performance Test (MPT) components. Used to assess legal reasoning capabilities across seven subject areas including Evidence, Torts, Constitutional Law, Contracts, Criminal Law and Procedure, Real Property, and Civil Procedure.","area":"Language & Knowledge","applicationDomains":["Law & Government"],"primaryDomain":"Law & Government","industrySectors":[],"capabilities":[],"topics":["Legal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/uniform-bar-exam","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5751aeb45604a440"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/uniform-bar-exam"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"uniform-bar-exam","url":"https://llm-stats.com/benchmarks/uniform-bar-exam","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["legal","reasoning"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_uniql_126b9bdb","familyId":"bmf_989990ea6746","name":"UniQL","oneLine":"UniQL is a human-verified benchmark for cross-dialect text-to-SQL, aligning 1,534 questions with executable SQL across 16 dialects, with dialect-specific evaluation tracks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08018","pdf":"https://arxiv.org/pdf/2606.08018","project":null,"code":"https://github.com/JerryGao818/UniQL","data":null,"hfPaper":"https://huggingface.co/papers/2606.08018"},"evidence":{"snippet":"We introduce UniQL, a human-verified benchmark for cross-dialect text-to-SQL evaluation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08018"},"ranking":{"90d":{"score":28,"rank":282,"coverage":0.55,"confidence":"Low"}},"description":"UniQL is a human-verified benchmark for cross-dialect text-to-SQL, aligning 1,534 questions with executable SQL across 16 dialects, with dialect-specific evaluation tracks.","whyItMatters":"Most text-to-SQL benchmarks are SQLite-only, but real systems use varied dialects; UniQL enables controlled evaluation of dialect generalization, highlighting transfer gaps.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3b4b4796ee2a186ddd8f0433b283c03b0f55191b2ee53590fee6ce7860ac6e9e"},"motivation":"Existing text-to-SQL benchmarks are largely centered on SQLite, making it difficult to evaluate whether models can generalize across heterogeneous SQL dialects.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08018","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"JerryGao818","organizationType":"community","sourceUrl":"https://github.com/JerryGao818/UniQL","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_unison_ffbd3e3c","familyId":"bmf_b3704083f1b6","name":"Unison","oneLine":"Evaluates unified multimodal models on joint understanding and generation across four dimensions: internal consistency, understanding-guided generation, generation-guided understanding, and mutual enhancement, using 2,169 task samples.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26984","pdf":"https://arxiv.org/pdf/2606.26984","project":null,"code":"https://github.com/FudanCVL/Unison","data":null,"hfPaper":"https://huggingface.co/papers/2606.26984"},"evidence":{"snippet":"To bridge this gap, we introduce Unison, a comprehensive benchmark comprising 2,169 high-quality unified task samples, designed to evaluate joint understanding and generation in unified multimodal models.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":14,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26984"},"ranking":{"90d":{"score":43,"rank":131,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates unified multimodal models on joint understanding and generation across four dimensions: internal consistency, understanding-guided generation, generation-guided understanding, and mutual enhancement, using 2,169 task samples.","whyItMatters":"Existing benchmarks assess understanding and generation in isolation, missing the synergy between them. This benchmark provides a diagnostic and human-aligned evaluation to measure and compare integrated capabilities, aiding model development.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"702c52da8df55316ee62da579bb57a5752fab832cf84ff1bbd8436c3b79ec1e0"},"motivation":"Unified multimodal models capable of both understanding and generation have achieved remarkable strides.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.26984","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"FudanCVL","organizationType":"academic-lab","sourceUrl":"https://github.com/FudanCVL/Unison","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_uoj-bench_c2013a7a","familyId":"bmf_e38a83758c16","name":"UOJ-Bench","oneLine":"UOJ-Bench is a benchmark for evaluating LLMs in code generation, hacking, and repair, built from real-world submissions on the Universal Online Judge and evaluated through UOJ's judging infrastructure.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation"],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.12864","pdf":"https://arxiv.org/pdf/2606.12864","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.12864"},"evidence":{"snippet":"In this work, we introduce UOJ-Bench, a benchmark designed to evaluate not only the problem-solving ability of LLMs, but also their ability to identify errors in human-written code -- a crucial educational activity traditionally supported by running test cases over online judge systems.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.12864"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"UOJ-Bench is a benchmark for evaluating LLMs in code generation, hacking, and repair, built from real-world submissions on the Universal Online Judge and evaluated through UOJ's judging infrastructure.","whyItMatters":"UOJ-Bench extends beyond problem-solving to include identifying errors in human code, a critical educational activity. It provides a realistic setting for assessing LLM support in learning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2f39985c58bc9171a708ac301f999dd18b2e1ba063f6730c1b45475f188de916"},"motivation":"Despite strong performance in competitive programming, the role of Large Language Models (LLMs) in supporting human learning in the same setting remains largely unexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.12864","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_urbanwell_81901465","familyId":"bmf_a905172c1364","name":"UrbanWell","oneLine":"UrbanWell provides a dataset and evaluation protocol for assessing spatio-temporal reasoning in multimodal large language models using satellite and street view imagery across 38 cities, covering environmental, accessibility, urban form, vitality, and subjective perception indicators with tasks in static prediction, forecasting, and trend classification.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.15890","pdf":"https://arxiv.org/pdf/2606.15890","project":null,"code":"https://github.com/axin1301/UrbanWell-Benchmark","data":null,"hfPaper":"https://huggingface.co/papers/2606.15890"},"evidence":{"snippet":"We introduce UrbanWell, a large-scale benchmark designed to systematically evaluate the spatio-temporal reasoning capabilities of MLLMs for urban wellbeing analytics through joint modeling of satellite and street view imagery.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.15890"},"ranking":{"90d":{"score":30,"rank":239,"coverage":0.7,"confidence":"Medium"}},"description":"UrbanWell provides a dataset and evaluation protocol for assessing spatio-temporal reasoning in multimodal large language models using satellite and street view imagery across 38 cities, covering environmental, accessibility, urban form, vitality, and subjective perception indicators with tasks in static prediction, forecasting, and trend classification.","whyItMatters":"UrbanWell addresses the lack of standardized benchmarks for multimodal urban wellbeing analytics, enabling consistent comparison of MLLMs on tasks requiring joint spatial and temporal understanding. This supports progress in urban intelligence applications such as planning and policy evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"38d41b8df9c86e25edea6a2ba6336fb02203463e7bf4f1b501b2881ba9dd8ec0"},"motivation":"Understanding urban wellbeing from multimodal data requires integrating heterogeneous spatial and temporal signals, posing significant challenges for current multimodal large language models (MLLMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"KDD Datasets and Benchmarks Track 2026","evidence":"accepted by KDD Datasets and Benchmarks Track 2026","evidenceUrl":"https://arxiv.org/abs/2606.15890","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"KDD Datasets and Benchmarks Track 2026","reviewStatus":"accepted","decisionRaw":"accepted by KDD Datasets and Benchmarks Track 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.15890","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"accepted by KDD Datasets and Benchmarks Track 2026","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_urdummlu_7d4881d1","familyId":"bmf_60682f36bc94","name":"UrduMMLU","oneLine":"UrduMMLU evaluates Urdu language understanding through 26,431 multiple-choice questions across 26 subjects and five domains, sourced from native educational materials. Accuracy under zero-shot and few-shot prompting protocols serves as the primary scoring metric.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07167","pdf":"https://arxiv.org/pdf/2606.07167","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07167"},"evidence":{"snippet":"We introduce UrduMMLU, a benchmark of 26,431 Urdu MCQs across 26 subjects and five domains, collected from native Urdu MCQ banks and public examination PDFs.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07167"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"UrduMMLU evaluates Urdu language understanding through 26,431 multiple-choice questions across 26 subjects and five domains, sourced from native educational materials. Accuracy under zero-shot and few-shot prompting protocols serves as the primary scoring metric.","whyItMatters":"Urdu, spoken by over 230 million people, lacks broad MMLU-style evaluation from native sources. UrduMMLU addresses this gap by testing models on region-specific knowledge, revealing uneven performance across subjects and providing a public benchmark for comparing LLMs on Urdu understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"36a90b37cdc68070a3b03bfc002d161b1ac854b5dd7d2d81a60d0358fd947750"},"motivation":"Meaningful multilingual evaluation must test models in the target language and educational context.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07167","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_f98a3c865f045fb7","familyId":"catalog_family_f98a3c865f045fb7","name":"USAMO 2026","oneLine":"USAMO 2026 evaluates models on the six problems from the 2026 United States of America Mathematical Olympiad, requiring rigorous proof-based reasoning. Following the MathArena methodology, proofs are graded against human expert rubrics by dual strong judge models, with the minimum of the two scores taken as final. Maximum score is 42 points.","description":"USAMO 2026 evaluates models on the six problems from the 2026 United States of America Mathematical Olympiad, requiring rigorous proof-based reasoning. Following the MathArena methodology, proofs are graded against human expert rubrics by dual strong judge models, with the minimum of the two scores taken as final. Maximum score is 42 points.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.maa.org/math-competitions/usamo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f98a3c865f045fb7"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/usamo2026"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/usamo-2026"}],"catalogSources":[{"catalog":"benchlm","sourceId":"usamo2026","url":"https://benchlm.ai/benchmarks/usamo2026","paperUrl":"https://www.maa.org/math-competitions/usamo","year":"2026","fullName":"United States of America Mathematical Olympiad 2026","format":"Mathematical proof construction","tasks":"6 proof-based problems","successorKey":null},{"catalog":"llm-stats","sourceId":"usamo-2026","url":"https://llm-stats.com/benchmarks/usamo-2026","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"catalog_2938cd779ea308ef","familyId":"catalog_family_2938cd779ea308ef","name":"USAMO25","oneLine":"The 2025 United States of America Mathematical Olympiad (USAMO) benchmark consists of six challenging mathematical problems requiring rigorous proof-based reasoning. USAMO is the most prestigious high school mathematics competition in the United States, serving as the final round of the American Mathematics Competitions series. This benchmark evaluates models on mathematical problem-solving capabilities beyond simple numerical computation, focusing on formal mathematical reasoning and proof generation.","description":"The 2025 United States of America Mathematical Olympiad (USAMO) benchmark consists of six challenging mathematical problems requiring rigorous proof-based reasoning. USAMO is the most prestigious high school mathematics competition in the United States, serving as the final round of the American Mathematics Competitions series. This benchmark evaluates models on mathematical problem-solving capabilities beyond simple numerical computation, focusing on formal mathematical reasoning and proof generation.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Math","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/usamo25","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2938cd779ea308ef"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/usamo25"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"usamo25","url":"https://llm-stats.com/benchmarks/usamo25","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["math","reasoning"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_usertoolbench_99873053","familyId":"bmf_2846551f77dd","name":"UserToolBench","oneLine":"UserToolBench evaluates personalized decision making in tool-use LLMs through inference of latent user preferences, clarification need, and user-aligned tool-call trajectories, built from privacy-sanitized interaction traces with 10 user profiles, 36 tool sets, and 1,065 turns.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.10042","pdf":"https://arxiv.org/pdf/2608.10042","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.10042"},"evidence":{"snippet":"We introduce UserToolBench , a benchmark for personalized decision making in tool-use LLMs.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10042"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"UserToolBench evaluates personalized decision making in tool-use LLMs through inference of latent user preferences, clarification need, and user-aligned tool-call trajectories, built from privacy-sanitized interaction traces with 10 user profiles, 36 tool sets, and 1,065 turns.","whyItMatters":"Current personalization benchmarks focus on surface-level style imitation or response personalization, not whether models make correct decisions for the user. UserToolBench addresses this gap by emphasizing decision quality, which is critical for real-world delegation tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e8ada43b31b6212f475fea541069db0033ca0cfc6d6846cc0ebc3a7528f1eb01"},"motivation":"Tool-use LLMs are increasingly asked to act on users' behalf, but existing benchmarks usually focus on profile recall, style imitation, generic tool use, or response-level personalization.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10042","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_uxbench_0fa77b90","familyId":"bmf_8d180072bdf4","name":"UXBench","oneLine":"UXBench evaluates LLM-generated UX critiques through local web fixtures, coverage-gated exploration, and a downstream repair agent, measuring report actionability across seven rubric dimensions.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-15","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.16262","pdf":"https://arxiv.org/pdf/2606.16262","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.16262"},"evidence":{"snippet":"We introduce UXBench, a benchmark for evaluating LLMs as interaction-grounded UX judges.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.16262"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"UXBench evaluates LLM-generated UX critiques through local web fixtures, coverage-gated exploration, and a downstream repair agent, measuring report actionability across seven rubric dimensions.","whyItMatters":"There is a need for a controlled evaluation of UX critique reliability and actionability across product surfaces, but UXBench currently lacks a public release path or ongoing scoring service, limiting its standalone comparison value.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c63b53bd30985886c51073e1def18aa13779f17349489f89d391f905ad7fb9c7"},"motivation":"Large language models (LLMs) are increasingly deployed as UX judges that inspect interfaces, diagnose usability problems, and propose repairs.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.16262","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_uxbench_9ba76902","familyId":"bmf_8d180072bdf4","name":"UXBench","oneLine":"UXBench evaluates user experience in AI assistants through three tasks: UX Judge (binary classification of response quality), UX Eval (response generation), and UX Recovery (repairing failed interactions). The dataset contains 7,400 test instances from 70K+ real interaction logs, covering 8 scenarios and 83 domains. Scoring uses accuracy for Judge and GRM-rated quality for generation tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09570","pdf":"https://arxiv.org/pdf/2606.09570","project":null,"code":"https://github.com/mengze-hong/UXBench","data":null,"hfPaper":"https://huggingface.co/papers/2606.09570"},"evidence":{"snippet":"We present UXBench, the first user-centric benchmark grounded in real user feedback signals for evaluating preference alignment and dialogue generation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":4,"hfDailySubmittedAt":null,"githubStars":15,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.09570"},"ranking":{"90d":{"score":40,"rank":151,"coverage":0.7,"confidence":"Medium"}},"description":"UXBench evaluates user experience in AI assistants through three tasks: UX Judge (binary classification of response quality), UX Eval (response generation), and UX Recovery (repairing failed interactions). The dataset contains 7,400 test instances from 70K+ real interaction logs, covering 8 scenarios and 83 domains. Scoring uses accuracy for Judge and GRM-rated quality for generation tasks.","whyItMatters":"UXBench addresses the gap in evaluating AI assistants beyond raw capability, focusing on user-perceived utility and preference alignment. It provides a structured way to measure how well models understand and improve user experience, offering practical value for developing assistants that better satisfy real users.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"189775d557918107eabb50b66a7917c5b1edc6e09eed26da6dc9b5754ef0339c"},"motivation":"As AI assistants serve millions of users daily, evaluating user experience (UX) beyond general model capability has become increasingly important.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09570","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_4c94485e0c21ae6c","familyId":"catalog_family_4c94485e0c21ae6c","name":"V*","oneLine":"A visual reasoning benchmark evaluating multimodal inference under challenging spatial and grounded tasks.","description":"A visual reasoning benchmark evaluating multimodal inference under challenging spatial and grounded tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4c94485e0c21ae6c"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/vstar"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/v-star"}],"catalogSources":[{"catalog":"benchlm","sourceId":"vStar","url":"https://benchlm.ai/benchmarks/vstar","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"V*","format":"Vision-centric reasoning benchmark","tasks":"Frontier multimodal reasoning tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"v-star","url":"https://llm-stats.com/benchmarks/v-star","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","vision"],"catalogModelCount":7,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_v2v-bench_86c97c1d","familyId":"bmf_6f509f52ee80","name":"V2V-Bench","oneLine":"V2V-Bench evaluates video-to-video generation models across 11 dimensions in five categories: temporal alignment, structural fidelity, transformation quality, video quality, and semantic alignment. It pairs source videos with editing tasks and scores models on these dimensions.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.05665","pdf":"https://arxiv.org/pdf/2606.05665","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.05665"},"evidence":{"snippet":"We introduce V2V-Bench, a 11-dimension benchmark organized into five categories: temporal alignment, structural fidelity, transformation quality, video quality, and semantic alignment.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.05665"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"V2V-Bench evaluates video-to-video generation models across 11 dimensions in five categories: temporal alignment, structural fidelity, transformation quality, video quality, and semantic alignment. It pairs source videos with editing tasks and scores models on these dimensions.","whyItMatters":"Existing T2V and I2V metrics do not capture the joint requirements of instruction following and frame-level correspondence in V2V generation. A dedicated benchmark with human-correlated scoring could support model selection for V2V applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"edc99abe4a50f224bdc676a072a8bb8d87646cfc5062ceff8091038fb55b60a5"},"motivation":"Video-to-video (V2V) generation is difficult to evaluate because outputs must both follow editing instructions and preserve frame-level correspondence with the source video, which existing T2V and I2V metrics do not capture.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ICML 2026 workshop","evidence":"Accepted at ICML 2026 workshop","evidenceUrl":"https://arxiv.org/abs/2606.05665","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ICML 2026 workshop","reviewStatus":"accepted","decisionRaw":"Accepted at ICML 2026 workshop","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.05665","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at ICML 2026 workshop","level":"author-claim"}]}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_va-judger-bench_7e8bd378","familyId":"bmf_52e48663243f","name":"VA-Judger-Bench","oneLine":"VA-Judger-Bench evaluates reward models for joint video-audio generation. It contains paired comparisons of generated video-audio samples with human preference labels, covering in-domain and out-of-domain model outputs.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-19","firstSeenAt":"2026-08-20","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.18607","pdf":"https://arxiv.org/pdf/2608.18607","project":null,"code":"https://github.com/ShareLab-SII/VA-Judger","data":null,"hfPaper":null},"evidence":{"snippet":"We also introduce the VA-Judger-Bench benchmark with both in-domain and out-of-domain model comparisons to evaluate whether reward models truly align with human preferences.","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence","public artifact URL","named entity is a model, method, framework, or dataset used in evaluation"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":11,"hfDailySubmittedAt":"2026-08-20T00:00:00.000Z","githubStars":57,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.18607"},"ranking":{"30d":{"score":54,"rank":15,"coverage":0.85,"confidence":"High"},"90d":{"score":52,"rank":54,"coverage":0.7,"confidence":"Medium"}},"description":"VA-Judger-Bench evaluates reward models for joint video-audio generation. It contains paired comparisons of generated video-audio samples with human preference labels, covering in-domain and out-of-domain model outputs.","whyItMatters":"Existing metrics evaluate quality dimensions separately, missing semantic and temporal coherence. VA-Judger-Bench provides a benchmark to assess whether reward models align with human preferences for joint video-audio generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f04b475f5407961dea11eea659358dca79197764b37006841a6c009cabb9b383"},"motivation":"Using reinforcement learning to post-train joint video-audio generation models requires a reward signal.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-20T09:45:58.504563Z","model":"deepseek-v4-flash","decisionReason":"VA-Judger-Bench is a named benchmark with a defined evaluation protocol, publicly released data/checkpoints, and clear scoring contract for comparing reward models."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.18607","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-20T09:44:47.308583Z"},"evaluationMode":"public_reusable","publishers":[{"name":"ShareLab-SII","organizationType":"academic-lab","sourceUrl":"https://github.com/ShareLab-SII/VA-Judger","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vakra-evaluating-multi-hop-reasoning-acros_7a94433a","familyId":"bmf_c97fe73eec92","name":"VAKRA","oneLine":"Evaluates multi-hop tool-use agents across live executable APIs and retrieval, scoring via policy adherence, exact match, and groundedness judges.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Information retrieval"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.12282","pdf":"https://arxiv.org/pdf/2608.12282","project":null,"code":"https://github.com/IBM/VAKRA","data":"https://huggingface.co/datasets/ibm-research/VAKRA","hfPaper":null},"evidence":{"snippet":"We introduce VAKRA (e\\textbf{V}aluating \\textbf{A}PI and \\textbf{K}nowledge \\textbf{R}etrieval \\textbf{A}gents), a benchmark of over $8{,}000$ executable APIs across $62$ domains with tasks spanning three settings of increasing difficulty: diverse API interaction styles, multi-hop reasoning over structured APIs, and multi-source reasoning with natural-language tool-use policy constraints.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":65,"githubScope":"benchmark_repo","hfDatasetDownloads":1758,"hfDatasetLikes":46},"source":{"type":"arxiv","id":"2608.12282"},"ranking":{"30d":{"score":61,"rank":9,"coverage":0.7,"confidence":"Medium","datasetDownloadRank":5,"datasetRankPopulation":30},"90d":{"score":57,"rank":27,"coverage":0.85,"confidence":"High","datasetDownloadRank":10,"datasetRankPopulation":66}},"description":"Evaluates multi-hop tool-use agents across live executable APIs and retrieval, scoring via policy adherence, exact match, and groundedness judges.","whyItMatters":"Enterprise agent evaluations need compositional multi-source reasoning with policy constraints; VAKRA offers a public leaderboard and reproducible harness.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"d36f88872a27bbf0d65acc7409d419ff1f13373b2e033dd04584754444c43c67"},"motivation":"Agents deployed in enterprise settings must reason across structured APIs and document collections, yet existing benchmarks evaluate these capabilities in isolation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"VAKRA is formally named, has a leaderboard, code, and dataset links indicating an ongoing submission-based benchmark.","canonicalNameSource":"paper_title","canonicalNameEvidence":"VAKRA: Evaluating Multi-Hop Reasoning Across APIs and Retrieval Under Tool-Use Policies"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.12282","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":74,"confidence":"Medium","horizon":"7d","reason":"VAKRA combines over 8,000 APIs, multi-hop reasoning, and policy constraints with a live leaderboard, broadening its appeal to agent evaluation researchers."},"evaluationMode":"score_submission","publishers":[{"name":"IBM Research","organizationType":"company-research-lab","sourceUrl":"https://github.com/IBM/VAKRA","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Knowledge & Reasoning","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_bca92bae49afc9cf","familyId":"catalog_family_bca92bae49afc9cf","name":"Vals GPQA Diamond mirror","oneLine":"Vals AI hosted GPQA Diamond view with few-shot and zero-shot chain-of-thought task splits.","description":"Vals AI hosted GPQA Diamond view with few-shot and zero-shot chain-of-thought task splits.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/gpqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bca92bae49afc9cf"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsgpqadiamond"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsGpqaDiamond","url":"https://benchlm.ai/benchmarks/valsgpqadiamond","paperUrl":"https://www.vals.ai/benchmarks/gpqa","year":"2026","fullName":"Vals-hosted GPQA Diamond mirror","format":"Accuracy score","tasks":"GPQA Diamond task splits","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_770df60f96365c45","familyId":"catalog_family_770df60f96365c45","name":"Vals Index","oneLine":"Vals AI composite benchmark across professional finance, coding, modeling, and legal-work tasks, including Finance Agent v2, EMB, Terminal-Bench 2.1, Vibe Code Bench, Code Migration, Legal Research, and HLAB.","description":"Vals AI composite benchmark across professional finance, coding, modeling, and legal-work tasks, including Finance Agent v2, EMB, Terminal-Bench 2.1, Vibe Code Bench, Code Migration, Legal Research, and HLAB.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/vals_index","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_770df60f96365c45"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsindex"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsIndex","url":"https://benchlm.ai/benchmarks/valsindex","paperUrl":"https://www.vals.ai/benchmarks/vals_index","year":"2026","fullName":"Vals Index v2","format":"Composite score","tasks":"Finance, coding, spreadsheet modeling, code migration, and legal-work components","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_30b5602f6ab7a08d","familyId":"catalog_family_30b5602f6ab7a08d","name":"Vals LiveCodeBench mirror","oneLine":"Vals AI implementation of LiveCodeBench with easy, medium, and hard task splits.","description":"Vals AI implementation of LiveCodeBench with easy, medium, and hard task splits.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/lcb","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_30b5602f6ab7a08d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valslivecodebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsLiveCodeBench","url":"https://benchlm.ai/benchmarks/valslivecodebench","paperUrl":"https://www.vals.ai/benchmarks/lcb","year":"2026","fullName":"Vals-hosted LiveCodeBench mirror","format":"Accuracy score","tasks":"Coding problem difficulty splits","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_52b4bfcdd9eda584","familyId":"catalog_family_52b4bfcdd9eda584","name":"Vals MMLU-Pro mirror","oneLine":"Vals AI hosted MMLU-Pro view with subject-level task splits.","description":"Vals AI hosted MMLU-Pro view with subject-level task splits.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/mmlu_pro","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_52b4bfcdd9eda584"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsmmlupro"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsMmluPro","url":"https://benchlm.ai/benchmarks/valsmmlupro","paperUrl":"https://www.vals.ai/benchmarks/mmlu_pro","year":"2026","fullName":"Vals-hosted MMLU-Pro mirror","format":"Accuracy score","tasks":"MMLU-Pro subject splits","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_cef967a46cf0f149","familyId":"catalog_family_cef967a46cf0f149","name":"Vals Multimodal Index","oneLine":"Vals AI multimodal composite across finance, coding, education, and mortgage-tax task families.","description":"Vals AI multimodal composite across finance, coding, education, and mortgage-tax task families.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/vals_multimodal_index","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_cef967a46cf0f149"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsmultimodalindex"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsMultimodalIndex","url":"https://benchlm.ai/benchmarks/valsmultimodalindex","paperUrl":"https://www.vals.ai/benchmarks/vals_multimodal_index","year":"2026","fullName":"Vals Multimodal Index v1.2","format":"Composite score","tasks":"Finance, coding, education, and mortgage-tax components","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_2416de0eb1cf6a30","familyId":"catalog_family_2416de0eb1cf6a30","name":"Vals SWE-bench mirror","oneLine":"Vals AI hosted SWE-bench view for solving production software engineering tasks.","description":"Vals AI hosted SWE-bench view for solving production software engineering tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/swebench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2416de0eb1cf6a30"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsswebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsSweBench","url":"https://benchlm.ai/benchmarks/valsswebench","paperUrl":"https://www.vals.ai/benchmarks/swebench","year":"2026","fullName":"Vals-hosted SWE-bench mirror","format":"Accuracy score","tasks":"Software engineering issue-resolution tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_7b3445b405541d0e","familyId":"catalog_family_7b3445b405541d0e","name":"Vals Terminal-Bench 1.0 mirror","oneLine":"A Vals-hosted view of Terminal-Bench 1.0 with easy, medium, and hard task splits.","description":"A Vals-hosted view of Terminal-Bench 1.0 with easy, medium, and hard task splits.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/terminal-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7b3445b405541d0e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsterminalbench1"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsTerminalBench1","url":"https://benchlm.ai/benchmarks/valsterminalbench1","paperUrl":"https://www.vals.ai/benchmarks/terminal-bench","year":"2026","fullName":"Vals-hosted Terminal-Bench 1.0 mirror","format":"Accuracy score","tasks":"Terminal tasks split by easy, medium, and hard difficulty","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_79b6baccef6dfb22","familyId":"catalog_family_79b6baccef6dfb22","name":"Vals Terminal-Bench 2.0 mirror","oneLine":"Vals AI hosted Terminal-Bench 2.0 view with easy, medium, and hard task splits.","description":"Vals AI hosted Terminal-Bench 2.0 view with easy, medium, and hard task splits.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/terminal-bench-2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_79b6baccef6dfb22"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valsterminalbench2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsTerminalBench2","url":"https://benchlm.ai/benchmarks/valsterminalbench2","paperUrl":"https://www.vals.ai/benchmarks/terminal-bench-2","year":"2026","fullName":"Vals-hosted Terminal-Bench 2.0 mirror","format":"Accuracy score","tasks":"Terminal task difficulty splits","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_vamos-bench_83d46046","familyId":"bmf_e925739160ae","name":"VAmoS Bench","oneLine":"VAmoS Bench evaluates complete voice-agent systems in a stateful customer-support task, with 100 scenarios, a simulated caller, real SQL tools, and binary assertions assessed against full interaction traces.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.27453","pdf":"https://arxiv.org/pdf/2607.27453","project":null,"code":"https://github.com/veris-ai/riley-agent","data":null,"hfPaper":"https://huggingface.co/papers/2607.27453"},"evidence":{"snippet":"To address this gap, we introduce VAmoS Bench, the Voice Agent Simulation Bench.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":16,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.27453"},"ranking":{"90d":{"score":44,"rank":120,"coverage":0.55,"confidence":"Low"}},"description":"VAmoS Bench evaluates complete voice-agent systems in a stateful customer-support task, with 100 scenarios, a simulated caller, real SQL tools, and binary assertions assessed against full interaction traces.","whyItMatters":"It measures end-to-end call containment and correct backend mutations, going beyond component metrics to capture task-level correctness in realistic scenarios.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7f0d7e4077e7b8b6c367032c3d08c46f652f9b532fc516ef18eeac0ab9c02205"},"motivation":"Production voice agents span cascaded, speech-to-speech, and hybrid architectures.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.27453","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Veris AI","organizationType":"company-research-lab","sourceUrl":"https://github.com/veris-ai/riley-agent","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vanillabench_f1c1ed7c","familyId":"bmf_4e12120315f2","name":"VanillaBench","oneLine":"VanillaBench evaluates the clean accuracy gap between adversarially trained models and vanilla (non-robust) reference models across four threat models. It defines a protocol for comparing robustness-accuracy trade-offs using model accuracy on standard benchmarks.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Robustness"],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-14","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.12545","pdf":"https://arxiv.org/pdf/2607.12545","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.12545"},"evidence":{"snippet":"We introduce VanillaBench, a systematic benchmark that makes this gap explicit.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.12545"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"VanillaBench evaluates the clean accuracy gap between adversarially trained models and vanilla (non-robust) reference models across four threat models. It defines a protocol for comparing robustness-accuracy trade-offs using model accuracy on standard benchmarks.","whyItMatters":"It addresses the evaluation gap in adversarial robustness research by quantifying the cost of robustness, providing practitioners with information needed to make informed deployment decisions regarding accuracy versus robustness.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bdea16c79e9813caaf260f629be8ef2decf83a33ad50b0de7b4d2c9373bb8e39"},"motivation":"Adversarial robustness research has produced hundreds of defended models over the past decade, yet the literature almost universally reports robustness results in isolation: standard (clean) accuracy and adversarial accuracy of the robust model are shown, but the gap to the corresponding vanilla model is rarely quantified.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.12545","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_varm-bench_2f35ad2f","familyId":"bmf_93360e9b8769","name":"VARM-Bench","oneLine":"VARM-Bench evaluates verifiable structured reasoning in Chinese abusive-speech moderation. It uses field-anchored chain-of-thought rationales with six decision fields, and a deterministic protocol assessing field correctness, alignment, output validity, and record errors.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15600","pdf":"https://arxiv.org/pdf/2608.15600","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.15600"},"evidence":{"snippet":"We introduce VARM-Bench, a benchmark for field-anchored chain-of-thought rationales in Chinese abusive-speech moderation.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15600"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"VARM-Bench evaluates verifiable structured reasoning in Chinese abusive-speech moderation. It uses field-anchored chain-of-thought rationales with six decision fields, and a deterministic protocol assessing field correctness, alignment, output validity, and record errors.","whyItMatters":"Existing benchmarks support classification but not verifiable reasoning. VARM-Bench provides an auditable protocol for evaluating moderation rationales, revealing that strong label performance can conceal errors in complete records.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0e7cbac4d6c2f3731a5ce2ebf627f02213295f0839649ceaf5d7919c7bc8b219"},"motivation":"The widespread circulation of abusive online content has increased the need for reliable moderation of Chinese social-media text.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15600","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vast-benchmarking_a1aad559","familyId":"bmf_6f91daf2a194","name":"Vast Benchmarking","oneLine":"Measures GPU throughput, effective CPU concurrency, memory bandwidth, and disk speed of rented Vast.ai containers under a fixed wall-clock budget.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-23","recognitionConfidence":0.4,"links":{"report":"https://github.com/drivelineresearch/vast-benchmarking","pdf":null,"project":"https://cloud.vast.ai/?ref_id=77898","code":"https://github.com/drivelineresearch/vast-benchmarking","data":null,"hfPaper":null},"evidence":{"snippet":"vast-benchmarking Bounded Vast.ai GPU, effective-CPU, memory, and disk benchmark with a SQLite leaderboard benchmark computer-vision cuda flask pytorch sqlite vast-ai Vast Benchmarking Measure the hardware capacity a rented container can actually use.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:drivelineresearch/vast-benchmarking"},"ranking":{"30d":{"score":23,"rank":117,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":321,"coverage":0.55,"confidence":"Low"}},"description":"Measures GPU throughput, effective CPU concurrency, memory bandwidth, and disk speed of rented Vast.ai containers under a fixed wall-clock budget.","whyItMatters":"It gives renters a reproducible way to compare real container capacity beyond advertised core counts and prices.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"0bf3317c0b47e09d72fad5184b867fa0e0363005d4de2eb38df0995c3b0ac1b4"},"motivation":"vast-benchmarking Bounded Vast.ai GPU, effective-CPU, memory, and disk benchmark with a SQLite leaderboard benchmark computer-vision cuda flask pytorch sqlite vast-ai Vast Benchmarking Measure the hardware capacity a rented container can actually use.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/drivelineresearch/vast-benchmarking","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-23T05:46:57.857309Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"A bounded hardware benchmark tied to a popular GPU rental marketplace has practical appeal for the Vast.ai user community."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_e0956798955b98b1","familyId":"catalog_family_e0956798955b98b1","name":"VATEX","oneLine":"VaTeX: A Large-Scale, High-Quality Multilingual Dataset for Video-and-Language Research. Contains over 41,250 videos and 825,000 captions in both English and Chinese, with over 206,000 English-Chinese parallel translation pairs. Supports multilingual video captioning and video-guided machine translation tasks.","description":"VaTeX: A Large-Scale, High-Quality Multilingual Dataset for Video-and-Language Research. Contains over 41,250 videos and 825,000 captions in both English and Chinese, with over 206,000 English-Chinese parallel translation pairs. Supports multilingual video captioning and video-guided machine translation tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Multimodal","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vatex","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e0956798955b98b1"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vatex"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vatex","url":"https://llm-stats.com/benchmarks/vatex","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","multimodal","video","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vbvr-pro_41ab73db","familyId":"bmf_1235f75ea50f","name":"VBVR-Pro","oneLine":"Provides 300 procedurally generated tasks for native visual reasoning through generation, with verifiable reward scorers and controlled modality comparisons.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-26","firstSeenAt":"2026-08-29","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.26105","pdf":"https://arxiv.org/pdf/2608.26105","project":"https://video-reason.com/","code":"https://github.com/Video-Reason/VBVR-Pro","data":null,"hfPaper":"https://huggingface.co/papers/2608.26105"},"evidence":{"snippet":"In this work, we introduce VBVR-Pro, a closed-loop testbed that makes native visual reasoning through generation trainable, verifiable, optimizable, and experimentally controllable.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":255,"hfDailySubmittedAt":"2026-08-27T00:00:00.000Z","githubStars":23,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.26105"},"ranking":{"30d":{"score":63,"rank":8,"coverage":0.85,"confidence":"High"},"90d":{"score":54,"rank":43,"coverage":0.7,"confidence":"Medium"}},"description":"Provides 300 procedurally generated tasks for native visual reasoning through generation, with verifiable reward scorers and controlled modality comparisons.","whyItMatters":"Enables trainable, verifiable, and optimizable visual reasoning with strong transfer to external benchmarks.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"1656d9ec05f107727e7e71fa3a57f5e08c9a47b44a6f67b95d6f2ecf6cb82fc3"},"motivation":"Native visual reasoning treats visual generation as the medium of reasoning itself: visual states (i.e.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.26105","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"attentionForecast":{"score":20,"confidence":"Low","horizon":"7d","reason":"A scalable verifiable suite for native visual reasoning may intrigue researchers, but lack of linked artifacts limits immediate engagement."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vcifbench_33b2eb08","familyId":"bmf_6ff650c908db","name":"VCIFBench","oneLine":"Evaluates complex instruction following for video understanding across content, format, style, and structure constraints with 306 test instructions.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-03","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.04588","pdf":"https://arxiv.org/pdf/2606.04588","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.04588"},"evidence":{"snippet":"We introduce VCIFBench, a benchmark for evaluating complex instruction following in video understanding.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.04588"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates complex instruction following for video understanding across content, format, style, and structure constraints with 306 test instructions.","whyItMatters":"Could address the need for video benchmarks testing explicit output constraints, but lacks public evidence of reuse path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a0650f57d226b4afb6aa37efdb4be71bc755d1d86b26e2289def51ca7d3cbada"},"motivation":"Multimodal large language models have made rapid progress in video understanding, yet existing benchmarks largely rely on simple prompts and provide limited evidence about whether models can satisfy explicit output constraints.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.04588","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_76bd7780e0e532c3","familyId":"catalog_family_76bd7780e0e532c3","name":"VCR_en_easy","oneLine":"Visual Commonsense Reasoning (VCR) benchmark that tests higher-order cognition and commonsense reasoning beyond simple object recognition. Models must answer challenging questions about images and provide rationales justifying their answers. The benchmark measures the ability to infer people's actions, goals, and mental states from visual context.","description":"Visual Commonsense Reasoning (VCR) benchmark that tests higher-order cognition and commonsense reasoning beyond simple object recognition. Models must answer challenging questions about images and provide rationales justifying their answers. The benchmark measures the ability to infer people's actions, goals, and mental states from visual context.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vcr-en-easy","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_76bd7780e0e532c3"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vcr-en-easy"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vcr-en-easy","url":"https://llm-stats.com/benchmarks/vcr-en-easy","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vector-bench_6366f6e3","familyId":"bmf_628ccde1c0ca","name":"Vector-Bench","oneLine":"Vector-Bench evaluates instruction-based SVG code editing with 40 repair tasks, deterministic specification rewards, and metrics including repair progress and unintended change rate, across 34 model endpoints.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.19056","pdf":"https://arxiv.org/pdf/2607.19056","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.19056"},"evidence":{"snippet":"We introduce Vector-Bench, a compact, difficult benchmark of 40 SVG repair tasks.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.19056"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Vector-Bench evaluates instruction-based SVG code editing with 40 repair tasks, deterministic specification rewards, and metrics including repair progress and unintended change rate, across 34 model endpoints.","whyItMatters":"Vector editing fidelity is under-evaluated; this benchmark provides precise specification metrics, but public availability is unclear.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"4ea72572a5660a2b2fe8e58d6ae2518f9bbae1eae024169476b004d7d32104b2"},"motivation":"Instruction-based vector editing requires two capabilities: making a requested change and leaving everything else alone.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.19056","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_vehbench_bf410800","familyId":"bmf_06afd602ad70","name":"VEHBench","oneLine":"VEHBench is a diagnostic benchmark for LLM-assisted vibration energy harvester design, featuring 763 tasks across four design roles: specification triage, verifier-guided search, corrupted-state recovery, and policy-conditioned selection. It uses an analytical physical oracle for scoring.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-07-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.18181","pdf":"https://arxiv.org/pdf/2607.18181","project":null,"code":null,"data":"https://huggingface.co/datasets/AnonymousVehbench/vehbench","hfPaper":"https://huggingface.co/papers/2607.18181"},"evidence":{"snippet":"We introduce VEHBench, an engineering-native diagnostic benchmark for LLM-assisted VEH design, featuring 763 literature-grounded tasks scored by an analytical physical oracle.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":299,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2607.18181"},"ranking":{"90d":{"score":49,"rank":70,"coverage":0.3,"confidence":"Low","datasetDownloadRank":31,"datasetRankPopulation":66}},"description":"VEHBench is a diagnostic benchmark for LLM-assisted vibration energy harvester design, featuring 763 tasks across four design roles: specification triage, verifier-guided search, corrupted-state recovery, and policy-conditioned selection. It uses an analytical physical oracle for scoring.","whyItMatters":"The benchmark addresses the need for stage-local evaluation of LLMs in coupled physical design workflows, revealing stage-dependent capabilities and response-control patterns. It provides practical value for selecting and routing models in engineering applications, highlighting that no single model dominates all stages.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"61fb0c8f13e3ab2a6f85fbcdd64789d095d4786248f6dc1a188b312d6f16662d"},"motivation":"Battery-free Internet of Things (IoT) requires iterative design of vibration energy harvesters (VEHs) under coupled physical constraints, while LLMs are emerging as interface layers for engineering workflows.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.18181","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_33a7ccc91a5843b6","familyId":"catalog_family_33a7ccc91a5843b6","name":"Vending-Bench 2","oneLine":"Vending-Bench 2 tests longer horizon planning capabilities by evaluating how well AI models can manage a simulated vending machine business over extended periods. The benchmark measures a model's ability to maintain consistent tool usage and decision-making for a full simulated year of operation, driving higher returns without drifting off task.","description":"Vending-Bench 2 tests longer horizon planning capabilities by evaluating how well AI models can manage a simulated vending machine business over extended periods. The benchmark measures a model's ability to maintain consistent tool usage and decision-making for a full simulated year of operation, driving higher returns without drifting off task.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vending-bench-2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_33a7ccc91a5843b6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vending-bench-2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vending-bench-2","url":"https://llm-stats.com/benchmarks/vending-bench-2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","agents"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_vendorbench-100_8cfd26ec","familyId":"bmf_4b8f07b501cd","name":"VendorBench-100","oneLine":"Evaluates deepfake image detectors across three paradigms—commercial APIs, vision LLMs, and open-source detectors—on a fixed 100-image adversarial corpus. Uses a unified output schema and scores primarily by Matthews correlation coefficient with ROC-AUC.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.06254","pdf":"https://arxiv.org/pdf/2607.06254","project":null,"code":"https://github.com/sharayu-20/vendorbench-100","data":null,"hfPaper":"https://huggingface.co/papers/2607.06254"},"evidence":{"snippet":"We introduce VendorBench-100, a cross-paradigm benchmark that evaluates 36 representative models using a single adversarial 100-image corpus, a unified output schema, and a common evaluation framework.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06254"},"ranking":{"90d":{"score":28,"rank":275,"coverage":0.55,"confidence":"Low"}},"description":"Evaluates deepfake image detectors across three paradigms—commercial APIs, vision LLMs, and open-source detectors—on a fixed 100-image adversarial corpus. Uses a unified output schema and scores primarily by Matthews correlation coefficient with ROC-AUC.","whyItMatters":"Provides a common ground for comparing disparate detector types, addressing the lack of unified evaluation. Identifies metric correlation and calibration issues that matter for real-world deployment decisions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"bd1b8e98e3b7544d6982fbf9b2e6e71e9d058d53e37115bd6d2068ad8605b349"},"motivation":"Deepfake image detection is served by three fundamentally different paradigms - commercial APIs, zero-shot vision-language models (LLMs), and open-source detectors - that are rarely evaluated under a common protocol, making direct comparison difficult.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06254","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Sharayu Deshmukh","organizationType":"community","sourceUrl":"https://github.com/sharayu-20/vendorbench-100","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_veritrip_b9f0bf63","familyId":"bmf_61992c4b1df5","name":"VeriTrip","oneLine":"VeriTrip benchmarks travel planning agents on evidence-grounded reasoning over unstructured web corpora, with a verifiable knowledge base for cell-wise verification of factual reliability.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning","Robustness"],"topics":["Agents"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-27","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.28683","pdf":"https://arxiv.org/pdf/2605.28683","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.28683"},"evidence":{"snippet":"We introduce VeriTrip, a verifiable benchmark designed to meet the increasing demands for agent robustness and reliability.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.28683"},"ranking":{},"description":"VeriTrip benchmarks travel planning agents on evidence-grounded reasoning over unstructured web corpora, with a verifiable knowledge base for cell-wise verification of factual reliability.","whyItMatters":"Targets robustness of planning agents, but the benchmark's data and verification protocol are not described with public artifacts, limiting its standalone use.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c83e7335268ac903efbe5d6bd90faebf5fedf582be1afb835d97f36efa7ffe92"},"motivation":"Existing benchmarks have laid the foundation for travel planning agents by establishing API-centric paradigms.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.28683","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_vero_3ed8d572","familyId":"bmf_5cca655ac072","name":"Vero","oneLine":"Vero evaluates repository-level verified code generation in Lean 4, with 43 multi-module instances, formal specifications, and proof-only and code-and-proof evaluation modes.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.LG"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-13","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.13522","pdf":"https://arxiv.org/pdf/2608.13522","project":null,"code":"https://github.com/sunblaze-ucb/vero","data":null,"hfPaper":"https://huggingface.co/papers/2608.13522"},"evidence":{"snippet":"To bridge this gap, we introduce Vero, the first benchmark to evaluate joint implementation and proof synthesis at the repository level.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":64,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.13522"},"ranking":{"30d":{"score":42,"rank":45,"coverage":0.85,"confidence":"High"},"90d":{"score":46,"rank":93,"coverage":0.7,"confidence":"Medium"}},"description":"Vero evaluates repository-level verified code generation in Lean 4, with 43 multi-module instances, formal specifications, and proof-only and code-and-proof evaluation modes.","whyItMatters":"It provides a testbed for measuring progress toward repository-scale verified software synthesis, where current agents fall short.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"388e10a5152ca39158cc75b598822a9c50dee75c9bfe67498740462ee46d4da5"},"motivation":"AI agents are increasingly used for programming, but do not provide any guarantee on the correctness of generated code.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.13522","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Sunblaze UCB","organizationType":"academic-lab","sourceUrl":"https://github.com/sunblaze-ucb/vero","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_verticue-bench_6f5c5a90","familyId":"bmf_fabfbf0a5160","name":"VertiCue-Bench","oneLine":"VertiCue-Bench evaluates multimodal large language models on geospatial reasoning using canopy height models to resolve 2D ambiguity in remote sensing natural scenes. It includes 1,534 instances across 17 tasks, testing height perception and semantic reasoning.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.25784","pdf":"https://arxiv.org/pdf/2605.25784","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.25784"},"evidence":{"snippet":"To address this gap, we introduce VertiCue-Bench, the first diagnostic benchmark for CHM-grounded geospatial reasoning.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25784"},"ranking":{},"description":"VertiCue-Bench evaluates multimodal large language models on geospatial reasoning using canopy height models to resolve 2D ambiguity in remote sensing natural scenes. It includes 1,534 instances across 17 tasks, testing height perception and semantic reasoning.","whyItMatters":"Current remote sensing benchmarks are mostly 2D-centric, failing in environments with spectral confusion. This benchmark addresses the gap of whether models can leverage vertical cues for semantic disambiguation, providing insights into geometry-to-semantics reasoning in MLLMs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e43330584dc237458c09f246ff95527fc3ae128e63af3a1dc78cbe71fd7b585d"},"motivation":"Multimodal Large Language Models (MLLMs) have recently shown promising progress in geospatial reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25784","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_ves-bench_b925ca24","familyId":"bmf_23230c43b2ce","name":"VES-Bench","oneLine":"Tests long-horizon video understanding with 600 Temporal Ordering and Event Counting questions, auditing whether decoded frames cover all evidence intervals.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-23","firstSeenAt":"2026-08-25","recognitionConfidence":0.95,"links":{"report":"http://arxiv.org/abs/2608.22516v1","pdf":"https://arxiv.org/pdf/2608.22516v1","project":"https://buaa-colalab.github.io/TRACE/","code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce VES-Bench, a 600-question benchmark of Temporal Ordering and Event Counting items over 348 public long videos.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22516"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Tests long-horizon video understanding with 600 Temporal Ordering and Event Counting questions, auditing whether decoded frames cover all evidence intervals.","whyItMatters":"Shifts evaluation from final answers to evidence coverage, revealing whether correct answers rest on complete observation of long videos.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"26230a94cb80423a590925fd33acf8f29148372b1dedaa39936d977aeee3866c"},"motivation":"A long-video answer is evidence-supported only when the frames decoded from the video cover every event the answer depends on.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with a clear evaluation protocol and a project page, though no dataset or code links are supplied in the input.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce VES-Bench, a 600-question benchmark of Temporal Ordering and Event Counting items"},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 Main Conference","evidence":"Accepted to EMNLP 2026 Main Conference. 19 pages, 5 figures, 6 tables. Project page: https://buaa-colalab.github.io/TRACE/","evidenceUrl":"http://arxiv.org/abs/2608.22516v1","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-25T17:37:25.889905Z"},"venueAttempts":[{"venueName":"EMNLP 2026 Main Conference","reviewStatus":"accepted","decisionRaw":"Accepted to EMNLP 2026 Main Conference. 19 pages, 5 figures, 6 tables. Project page: https://buaa-colalab.github.io/TRACE/","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"http://arxiv.org/abs/2608.22516v1","observedAt":"2026-08-25T17:37:25.889905Z","rawValue":"Accepted to EMNLP 2026 Main Conference. 19 pages, 5 figures, 6 tables. Project page: https://buaa-colalab.github.io/TRACE/","level":"author-claim"}]}],"attentionForecast":{"score":50,"confidence":"Low","horizon":"7d","reason":"Novel audit-based evaluation for long video understanding may interest the video-language community, but missing artifact links in the input reduce immediate spread."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception","Search & Retrieval"],"domainScope":"general"},{"id":"bm_vg-guibench_23eefff5","familyId":"bmf_42e90ac524ad","name":"VG-GUIBench","oneLine":"VG-GUI-Bench evaluates MLLM-based GUI agents on following video tutorials to complete interactive tasks, with 1,000 long-horizon test cases and four metrics.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.29445","pdf":"https://arxiv.org/pdf/2606.29445","project":"https://vg-gui-tasker.github.io/","code":"https://github.com/VG-GUI-TASKER/VG-GUI-TASKER","data":null,"hfPaper":"https://huggingface.co/papers/2606.29445"},"evidence":{"snippet":"To address this gap, we introduce VG-GUIBench (Video-Guided GUI Benchmark), a new benchmark designed to evaluate whether MLLM-based GUI agents can follow video tutorials to complete corresponding GUI interactive tasks.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":28,"hfDailySubmittedAt":"2026-06-30T00:00:00.000Z","githubStars":19,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.29445"},"ranking":{"90d":{"score":46,"rank":91,"coverage":0.7,"confidence":"Medium"}},"description":"VG-GUI-Bench evaluates MLLM-based GUI agents on following video tutorials to complete interactive tasks, with 1,000 long-horizon test cases and four metrics.","whyItMatters":"The benchmark addresses video-guided agentic tasks, complementing VideoQA benchmarks for procedural knowledge transfer.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"64a241c8ce1d20bf21e270316b808a1e681e5fe8998252866a35ceb5d8da392d"},"motivation":"Video understanding is a fundamental capability for multimodal intelligence, and recent Multimodal Large Language Models (MLLMs) have achieved remarkable performance on Video Question Answering (VideoQA) benchmarks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"ECCV 2026","evidence":"Accepted by ECCV 2026. Project Page: https://vg-gui-tasker.github.io/","evidenceUrl":"https://arxiv.org/abs/2606.29445","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"ECCV 2026","reviewStatus":"accepted","decisionRaw":"Accepted by ECCV 2026. Project Page: https://vg-gui-tasker.github.io/","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.29445","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by ECCV 2026. Project Page: https://vg-gui-tasker.github.io/","level":"author-claim"}]}],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_vga-benchv2_d6dfef90","familyId":"bmf_05bcf3702015","name":"VGA-BenchV2","oneLine":"Evaluates video generation quality and aesthetic value with 1,016 prompts, 60,000 videos, and 36,000 task-level annotations.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-08-26","firstSeenAt":"2026-08-27","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25452","pdf":"https://arxiv.org/pdf/2608.25452","project":null,"code":null,"data":"https://huggingface.co/datasets/BestiVictoryLab/VGA-Bench","hfPaper":null},"evidence":{"snippet":"We introduce VGA-BenchV2, an extended human-aligned benchmark and optimization framework for jointly evaluating and improving video generation quality and aesthetic value.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":90,"hfDatasetLikes":0},"source":{"type":"arxiv","id":"2608.25452"},"ranking":{"30d":{"score":50,"rank":30,"coverage":0.15,"confidence":"Low","datasetDownloadRank":20,"datasetRankPopulation":30},"90d":{"score":46,"rank":94,"coverage":0.3,"confidence":"Low","datasetDownloadRank":47,"datasetRankPopulation":66}},"description":"Evaluates video generation quality and aesthetic value with 1,016 prompts, 60,000 videos, and 36,000 task-level annotations.","whyItMatters":"Provides human-aligned evaluation and an evaluation-to-optimization pipeline, enabling reinforcement learning-based generator fine-tuning.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"c16349bf58c1f0be8d57b118c9f95e78b6e329ec85550753316b189862ad6481"},"motivation":"We introduce VGA-BenchV2, an extended human-aligned benchmark and optimization framework for jointly evaluating and improving video generation quality and aesthetic value.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:12:36.511123Z","model":"deepseek-v4-pro","decisionReason":"Includes a Hugging Face dataset with prompts, videos, and annotations, plus multiple evaluation dimensions and a clearly defined scoring contract.","canonicalNameSource":"paper_title","canonicalNameEvidence":"VGA-BenchV2: An Expanded Unified Benchmark and Multi-Model Framework for Evaluating Video Aesthetics and Generation Quality"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25452","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-27T04:12:10.575570Z"},"attentionForecast":{"score":25,"confidence":"Medium","horizon":"7d","reason":"IJCAI 2026 acceptance and public data release support moderate visibility in the video generation evaluation community."},"evaluationMode":"public_reusable","publishers":[{"name":"BestiVictoryLab","organizationType":"academic-lab","sourceUrl":"https://huggingface.co/datasets/BestiVictoryLab/VGA-Bench","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vgenst-bench_f8fb9973","familyId":"bmf_3236a25f97a3","name":"VGenST-Bench","oneLine":"VGenST-Bench evaluates spatio-temporal reasoning in multimodal large language models using 1,200 procedurally generated videos with controlled scene composition, camera trajectory, and reasoning targets. It covers 12 reasoning tasks across three spatial scales and 12 QA types across three reasoning levels, with multiple choice and open-ended question variants.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22570","pdf":"https://arxiv.org/pdf/2605.22570","project":"https://zinosii.github.io/VGenST-Bench/","code":"https://github.com/zinosii/VGenST-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2605.22570"},"evidence":{"snippet":"In this paper, we introduce VGenST-Bench, a video benchmark that employs generative models to actively synthesize highly controlled and diverse evaluation scenarios.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":24,"hfDailySubmittedAt":null,"githubStars":14,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.22570"},"ranking":{},"description":"VGenST-Bench evaluates spatio-temporal reasoning in multimodal large language models using 1,200 procedurally generated videos with controlled scene composition, camera trajectory, and reasoning targets. It covers 12 reasoning tasks across three spatial scales and 12 QA types across three reasoning levels, with multiple choice and open-ended question variants.","whyItMatters":"Existing video reasoning benchmarks rely on static or passively curated content, limiting fine-grained diagnosis. VGenST-Bench uses active synthesis to enable controlled and diverse evaluation of fine-grained spatio-temporal reasoning, supporting model comparison and targeted improvement in MLLMs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"da5d35794b1ed6f3b0129bc5f428a0f94f85a866a439b03f27a0decc4c7653cf"},"motivation":"Spatio-temporal reasoning is a core capability for Multimodal Large Language Models (MLLMs) operating in the real world.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.22570","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Sungkyunkwan University","organizationType":"academic-lab","sourceUrl":"https://zinosii.github.io/VGenST-Bench/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vgi-bench_4a5d68ac","familyId":"bmf_756186ceeff4","name":"VGI-BENCH","oneLine":"Contains 27 tasks and 810 instances organized by task domains and skill tags to evaluate visual reasoning capabilities of video generation models.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-08-20","firstSeenAt":"2026-08-21","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.19583","pdf":"https://arxiv.org/pdf/2608.19583","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"To this end, we introduce VGI-bench, containing 27 tasks and 810 instances, organized by a two-level taxonomy of task domains and skill tags for fine-grained evaluation of visual reasoning capabilities of video generation models.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.19583"},"ranking":{"30d":{"score":40,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Contains 27 tasks and 810 instances organized by task domains and skill tags to evaluate visual reasoning capabilities of video generation models.","whyItMatters":"Provides fine-grained evaluation of video generation models' visual reasoning, addressing input alignment, process validity, and task difficulty calibration.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-25T17:43:56.718736Z","inputHash":"cc2312da59effb404c76dffabac3e318f5329779269455ecdee62b961daa4485"},"motivation":"Recent studies suggest that video generation models can exhibit certain forms of zero-shot visual reasoning through generated frames.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-25T17:43:56.718736Z","model":"deepseek-v4-pro","decisionReason":"The paper introduces a formal benchmark with a clear taxonomy and evaluation criteria, and the website provides a reuse path for other teams.","canonicalNameSource":"paper_title","canonicalNameEvidence":"VGI-BENCH: Probing Visual Intelligence in Video Generation Models"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.19583","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-21T04:30:09.569171Z"},"attentionForecast":{"score":80,"confidence":"Medium","horizon":"7d","reason":"The benchmark targets a trending area of video generation and reports strong model performance gaps, likely to attract substantial research attention."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vhdlbench_8f92d226","familyId":"bmf_866c4b9ba052","name":"VHDLBench","oneLine":"VHDLSuite is a benchmark-centered infrastructure for VHDL generation evaluation, integrating automated benchmark synthesis, executable validation, and multi-model diagnostic analysis. It includes VHDLBench with over 200 VHDL problems with validated testbenches.","area":"Language & Knowledge","applicationDomains":["Industrial & Engineering"],"primaryDomain":"Industrial & Engineering","industrySectors":["Semiconductors"],"capabilities":[],"topics":["cs.AR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13735","pdf":"https://arxiv.org/pdf/2606.13735","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.13735"},"evidence":{"snippet":"Second, we introduce VHDLBench, a benchmark with over 200 VHDL problems with complete and validated testbenches across a wide range of complexity levels.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13735"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"VHDLSuite is a benchmark-centered infrastructure for VHDL generation evaluation, integrating automated benchmark synthesis, executable validation, and multi-model diagnostic analysis. It includes VHDLBench with over 200 VHDL problems with validated testbenches.","whyItMatters":"Evaluating LLM performance in VHDL generation is limited compared to Verilog. VHDLSuite provides a standardized pipeline for scalable VHDL evaluation, addressing distinct language characteristics.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d40def589edb329f931ed356414118606c52f8974e9d8862a8084b53a19f8cfe"},"motivation":"Large Language Models (LLM) have shown impressive capabilities in Register Transfer Level (RTL) code generation, particularly for Verilog.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13735","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_viabench_7ade7668","familyId":"bmf_3bab85546b0e","name":"VIABench","oneLine":"VIABench evaluates multimodal large language models on three tasks from first-person videos of visually impaired individuals: proactive reminder, visual question answering, and vision-guided interaction. It includes 761 videos, 46.9 hours, and 14,526 annotations, with protocols for online and offline settings.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.14660","pdf":"https://arxiv.org/pdf/2607.14660","project":null,"code":"https://github.com/MCG-NJU/VIABench","data":null,"hfPaper":"https://huggingface.co/papers/2607.14660"},"evidence":{"snippet":"To fill this gap, we introduce VIABench, a comprehensive video benchmark specifically designed to evaluate MLLMs in Visually Impaired Assistance scenarios using first-person videos recorded or shared by VIIs themselves.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":9,"hfDailySubmittedAt":null,"githubStars":5,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.14660"},"ranking":{"90d":{"score":35,"rank":202,"coverage":0.7,"confidence":"Medium"}},"description":"VIABench evaluates multimodal large language models on three tasks from first-person videos of visually impaired individuals: proactive reminder, visual question answering, and vision-guided interaction. It includes 761 videos, 46.9 hours, and 14,526 annotations, with protocols for online and offline settings.","whyItMatters":"General MLLMs are rarely tested for real-world assistance of blind users. VIABench focuses on tasks like anticipating navigation-critical events, which are underrepresented in existing benchmarks, providing a practical measure of model utility in assistive contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d1ad5b80f3052a709b1912c1c26133d7314ca31648473995fab20925fa140868"},"motivation":"Visually impaired individuals (VIIs) encounter significant daily challenges due to limited access to visual information.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.14660","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MCG-NJU","organizationType":"academic-lab","sourceUrl":"https://github.com/MCG-NJU/VIABench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vialectbench_c8a2ee22","familyId":"bmf_9a6b7205c2c0","name":"VialectBench","oneLine":"Evaluates LLM robustness to Vietnamese dialectal rewrites across emotion recognition, natural language inference, question answering, and multiple-choice QA, measuring performance degradation across six dialect groups.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.10414","pdf":"https://arxiv.org/pdf/2608.10414","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.10414"},"evidence":{"snippet":"We introduce VialectBench (Vietnamese Dialects Benchmarking), a controlled benchmark for testing whether model decisions remain stable across six Vietnamese dialect groups.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10414"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates LLM robustness to Vietnamese dialectal rewrites across emotion recognition, natural language inference, question answering, and multiple-choice QA, measuring performance degradation across six dialect groups.","whyItMatters":"Highlights that model performance on standard Vietnamese does not guarantee reliable behavior under regional variation, informing deployment decisions for Vietnamese-language applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"170093ed0814dd2d3c3463a8f2410637b47634d5d52ece9765daf3e6872738a5"},"motivation":"Large Language Models (LLMs) are typically evaluated on standard written Vietnamese, yet everyday communication frequently involves regional dialects that preserve meaning but differ in surface form.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10414","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_vibe_30ded12f","familyId":"bmf_8368ec5f1a15","name":"VIBE","oneLine":"VIBE is a benchmark for entity-centered affective profiling of LLM outputs in Valence-Arousal-Dominance space, introducing a measurement contract with scalar favorability, response-level and target-directed VAD, and an Affective Passport reporting format.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03810","pdf":"https://arxiv.org/pdf/2608.03810","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.03810"},"evidence":{"snippet":"We introduce VIBE, a benchmark for entity-centered affective profiling of LLM outputs in Valence-Arousal-Dominance (VAD) space.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03810"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"VIBE is a benchmark for entity-centered affective profiling of LLM outputs in Valence-Arousal-Dominance space, introducing a measurement contract with scalar favorability, response-level and target-directed VAD, and an Affective Passport reporting format.","whyItMatters":"Existing sentiment and emotion benchmarks do not combine target-directed VAD attribution with an explicit scorer contract and passport reporting. VIBE provides a standardized way to report affective profiles, supporting practice-oriented evaluation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ad711d62b543408e9ddd88bcf72ef384f476ef4e8c6b77ce19461653869a0fd4"},"motivation":"Large language models routinely describe socially salient targets, including political figures, countries, religions, organizations, historical events, and social groups, encoding affective framing alongside factual content: a target may appear favorable or threatening, calm or conflictual, powerful or vulnerable.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03810","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"vibe","url":"https://llm-stats.com/benchmarks/vibe","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0},{"id":"catalog_c5fd1801a2c3042f","familyId":"catalog_family_c5fd1801a2c3042f","name":"VIBE Android","oneLine":"VIBE benchmark subset for Android application generation","description":"VIBE benchmark subset for Android application generation","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vibe-android","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c5fd1801a2c3042f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vibe-android"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vibe-android","url":"https://llm-stats.com/benchmarks/vibe-android","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_c1d1fee0593d8c3d","familyId":"catalog_family_c1d1fee0593d8c3d","name":"VIBE Backend","oneLine":"VIBE benchmark subset for backend service generation","description":"VIBE benchmark subset for backend service generation","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vibe-backend","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c1d1fee0593d8c3d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vibe-backend"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vibe-backend","url":"https://llm-stats.com/benchmarks/vibe-backend","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_c823c6e321237a56","familyId":"catalog_family_c823c6e321237a56","name":"Vibe Code Bench","oneLine":"Vals.ai benchmark for evaluating whether models can build complete web applications from natural language specifications in a production-like development environment.","description":"Vals.ai benchmark for evaluating whether models can build complete web applications from natural language specifications in a production-like development environment.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/vibe-code","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c823c6e321237a56"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/vibecodebench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"vibeCodeBench","url":"https://benchlm.ai/benchmarks/vibecodebench","paperUrl":"https://www.vals.ai/benchmarks/vibe-code","year":"2026","fullName":"Vibe Code Bench v1.1","format":"Full-stack app implementation benchmark","tasks":"End-to-end web application builds","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_a13f25bf6c359279","familyId":"catalog_family_a13f25bf6c359279","name":"VIBE iOS","oneLine":"VIBE benchmark subset for iOS application generation","description":"VIBE benchmark subset for iOS application generation","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vibe-ios","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a13f25bf6c359279"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vibe-ios"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vibe-ios","url":"https://llm-stats.com/benchmarks/vibe-ios","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_0fa3c47d9e1f8121","familyId":"catalog_family_0fa3c47d9e1f8121","name":"VIBE Simulation","oneLine":"VIBE benchmark subset for simulation code generation","description":"VIBE benchmark subset for simulation code generation","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vibe-simulation","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_0fa3c47d9e1f8121"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vibe-simulation"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vibe-simulation","url":"https://llm-stats.com/benchmarks/vibe-simulation","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_e15ee86f46c26205","familyId":"catalog_family_e15ee86f46c26205","name":"VIBE V2","oneLine":"VIBE-V2 is an internal benchmark covering pure front-end and full-stack Web, Android, and iOS projects with build-from-scratch tasks. It uses an Agent-as-a-Verifier paradigm to automatically verify program interaction logic and visual output, scoring models through a unified pipeline that includes a requirement set, containerized deployment, and a dynamic interaction environment.","description":"VIBE-V2 is an internal benchmark covering pure front-end and full-stack Web, Android, and iOS projects with build-from-scratch tasks. It uses an Agent-as-a-Verifier paradigm to automatically verify program interaction logic and visual output, scoring models through a unified pipeline that includes a requirement set, containerized deployment, and a dynamic interaction environment.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/MiniMaxAI/MiniMax-M3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e15ee86f46c26205"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/vibev2"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vibe-v2"}],"catalogSources":[{"catalog":"benchlm","sourceId":"vibeV2","url":"https://benchlm.ai/benchmarks/vibev2","paperUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","year":"2026","fullName":"VIBE V2","format":"Task success rate","tasks":"End-to-end coding-agent tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"vibe-v2","url":"https://llm-stats.com/benchmarks/vibe-v2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","agents","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_58b102d24a7d44bc","familyId":"catalog_family_58b102d24a7d44bc","name":"VIBE Web","oneLine":"VIBE benchmark subset for web application generation","description":"VIBE benchmark subset for web application generation","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vibe-web","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_58b102d24a7d44bc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vibe-web"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vibe-web","url":"https://llm-stats.com/benchmarks/vibe-web","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_926ff83a1e9c46a4","familyId":"catalog_family_926ff83a1e9c46a4","name":"Vibe-Eval","oneLine":"VIBE-Eval is a hard evaluation suite for measuring progress of multimodal language models, consisting of 269 visual understanding prompts with gold-standard responses authored by experts. The benchmark has dual objectives: vibe checking multimodal chat models for day-to-day tasks and rigorously testing frontier models, with the hard set containing >50% questions that all frontier models answer incorrectly.","description":"VIBE-Eval is a hard evaluation suite for measuring progress of multimodal language models, consisting of 269 visual understanding prompts with gold-standard responses authored by experts. The benchmark has dual objectives: vibe checking multimodal chat models for day-to-day tasks and rigorously testing frontier models, with the hard set containing >50% questions that all frontier models answer incorrectly.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","General","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vibe-eval","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_926ff83a1e9c46a4"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vibe-eval"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vibe-eval","url":"https://llm-stats.com/benchmarks/vibe-eval","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","general","vision"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_dd66ecafddb88c16","familyId":"catalog_family_dd66ecafddb88c16","name":"VIBE-Pro","oneLine":"VIBE-Pro is an advanced version of the VIBE (Visual & Interactive Benchmark for Execution) benchmark that evaluates LLMs on professional-grade full-stack application development tasks. It measures model performance across complex real-world development scenarios including web, mobile, and backend applications with higher difficulty than the standard VIBE benchmark.","description":"VIBE-Pro is an advanced version of the VIBE (Visual & Interactive Benchmark for Execution) benchmark that evaluates LLMs on professional-grade full-stack application development tasks. It measures model performance across complex real-world development scenarios including web, mobile, and backend applications with higher difficulty than the standard VIBE benchmark.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.minimax.io/news/minimax-m27-en","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dd66ecafddb88c16"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/vibepro"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vibe-pro"}],"catalogSources":[{"catalog":"benchlm","sourceId":"vibePro","url":"https://benchlm.ai/benchmarks/vibepro","paperUrl":"https://www.minimax.io/news/minimax-m27-en","year":"2026","fullName":"VIBE-Pro","format":"Repository-level implementation benchmark","tasks":"Full project delivery tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"vibe-pro","url":"https://llm-stats.com/benchmarks/vibe-pro","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["coding","agents","code"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_vibelifebench_bf7626dc","familyId":"bmf_978e50cecb82","name":"VibeLifeBench","oneLine":"VibeLifeBench evaluates LLM agents on 200 long-horizon, multi-week tasks across 10 everyday-life domains in a simulated world of 22 mock services. Agents must manage silent world changes and implicit constraints, with fine-grained weighted checks on end state, timeliness, and constraint adherence.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.10875","pdf":"https://arxiv.org/pdf/2608.10875","project":null,"code":"https://github.com/evolvent-ai/VibeLifeBench","data":null,"hfPaper":"https://huggingface.co/papers/2608.10875"},"evidence":{"snippet":"We introduce VibeLifeBench, a benchmark of 200 long-horizon tasks across ten everyday-life domains.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":17,"hfDailySubmittedAt":"2026-08-12T00:00:00.000Z","githubStars":15,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10875"},"ranking":{"30d":{"score":46,"rank":39,"coverage":0.85,"confidence":"High"},"90d":{"score":43,"rank":125,"coverage":0.7,"confidence":"Medium"}},"description":"VibeLifeBench evaluates LLM agents on 200 long-horizon, multi-week tasks across 10 everyday-life domains in a simulated world of 22 mock services. Agents must manage silent world changes and implicit constraints, with fine-grained weighted checks on end state, timeliness, and constraint adherence.","whyItMatters":"Existing benchmarks focus on short, static tasks, leaving a gap in measuring long-horizon proactive assistance. VibeLifeBench provides a credible public evaluation for agents that must operate over weeks with changing environments, offering practical decision value for deploying personal assistants.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c993e63c4cfcbf56c87edd9bccb6ac81aa608560b84648e6773efb99ddfea1bd"},"motivation":"Large language model (LLM) agents are increasingly deployed as personal assistants.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10875","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Evolvent AI","organizationType":"company-research-lab","sourceUrl":"https://github.com/evolvent-ai/VibeLifeBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_video-ifbench_4fc7a0ab","familyId":"bmf_cff0b9680f66","name":"Video-IFBench","oneLine":"Evaluates instruction following of multimodal LLMs in video understanding with 1.5K samples across constraint categories.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-26","firstSeenAt":"2026-08-29","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25529","pdf":"https://arxiv.org/pdf/2608.25529","project":null,"code":"https://github.com/Alexios-hub/Video-IFBench","data":null,"hfPaper":"https://huggingface.co/papers/2608.25529"},"evidence":{"snippet":"To address this gap, we introduce Video-IFBench, a comprehensive benchmark for evaluating instruction following in video understanding, where models must satisfy diverse user-specified constraints, including those grounded in visual and audio content.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":16,"hfDailySubmittedAt":"2026-08-27T00:00:00.000Z","githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25529"},"ranking":{"30d":{"score":23,"rank":106,"coverage":0.85,"confidence":"High"},"90d":{"score":23,"rank":310,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates instruction following of multimodal LLMs in video understanding with 1.5K samples across constraint categories.","whyItMatters":"Addresses the underexplored area of instruction adherence in video understanding, where user-specified constraints are crucial.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"dd6e3a99912f549e400c2e2765a9d6888b5beafd1a804420beabda33caf84ae0"},"motivation":"Multimodal Large Language Models (MLLMs) have shown strong performance in video understanding.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"rule-auto-admitted","reviewedAt":"2026-08-31T04:05:24.949457Z","policy":"source-evidence-v2","decisionReason":"Automatically admitted from an explicit named benchmark release with evaluation-protocol evidence at or above the configured publication-confidence threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.25529","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"attentionForecast":{"score":10,"confidence":"Low","horizon":"7d","reason":"Without released artifacts, interest may remain limited to paper readers despite the benchmark's topical relevance."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"lib_video_mme","familyId":"family_video_mme","name":"Video-MME","oneLine":"Established benchmark family · Video Understanding.","area":"Video Understanding","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Video Understanding"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2405.21075","pdf":null,"project":"https://video-mme.github.io/home_page.html","code":"https://github.com/BradyFU/Video-MME","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_video_mme"},"ranking":{},"recordType":"family","aliases":["Video MME"],"sourceAttribution":[{"role":"official-project","url":"https://video-mme.github.io/home_page.html"}],"adoptionRefs":["google-gemini25"],"modelReportReferences":[{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"videoMme","url":"https://benchlm.ai/benchmarks/videomme","paperUrl":"https://mme-benchmark.github.io/","year":"2024","fullName":"Video-MME","format":"Video QA and analysis","tasks":"Video understanding","successorKey":null},{"catalog":"llm-stats","sourceId":"video-mme","url":"https://llm-stats.com/benchmarks/video-mme","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","vision"],"catalogModelCount":17,"catalogStarCount":0},{"id":"catalog_621de856e8fd1d1d","familyId":"catalog_family_621de856e8fd1d1d","name":"Video-MME (long, no subtitles)","oneLine":"Video-MME is the first-ever comprehensive evaluation benchmark for Multi-modal Large Language Models (MLLMs) in video analysis. This variant focuses on long-term videos (30min-60min) without subtitle inputs, testing robust contextual dynamics across 6 primary visual domains with 30 subfields including knowledge, film & television, sports competition, life record, and multilingual content.","description":"Video-MME is the first-ever comprehensive evaluation benchmark for Multi-modal Large Language Models (MLLMs) in video analysis. This variant focuses on long-term videos (30min-60min) without subtitle inputs, testing robust contextual dynamics across 6 primary visual domains with 30 subfields including knowledge, film & television, sports competition, life record, and multilingual content.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/video-mme-(long,-no-subtitles)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_621de856e8fd1d1d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/video-mme-(long,-no-subtitles)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"video-mme-(long,-no-subtitles)","url":"https://llm-stats.com/benchmarks/video-mme-(long,-no-subtitles)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","video","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_72b2f91549e21bfc","familyId":"catalog_family_72b2f91549e21bfc","name":"Video-MME (w/o subtitle)","oneLine":"A stricter Video-MME setting that removes subtitle help and tests video understanding from visual and audio context alone.","description":"A stricter Video-MME setting that removes subtitle help and tests video understanding from visual and audio context alone.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_72b2f91549e21bfc"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/videommenosub"}],"catalogSources":[{"catalog":"benchlm","sourceId":"videoMmeNoSub","url":"https://benchlm.ai/benchmarks/videommenosub","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"Video-MME without subtitle","format":"Video QA without subtitle context","tasks":"Video understanding","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_369d5330062373a8","familyId":"catalog_family_369d5330062373a8","name":"Video-MME (with subtitle)","oneLine":"A video understanding benchmark that allows subtitle access when answering multimodal questions about videos.","description":"A video understanding benchmark that allows subtitle access when answering multimodal questions about videos.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_369d5330062373a8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/videommewithsub"}],"catalogSources":[{"catalog":"benchlm","sourceId":"videoMmeWithSub","url":"https://benchlm.ai/benchmarks/videommewithsub","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"Video-MME with subtitle","format":"Video QA with subtitle context","tasks":"Video understanding","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_video-mme-logical_db483a80","familyId":"bmf_406a86346e24","name":"Video-MME-Logical","oneLine":"Evaluates video temporal-logical reasoning in multimodal LLMs across 25 fine-grained task categories, with difficulty-controlled final-answer scoring and intermediate-state diagnostics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.27828","pdf":"https://arxiv.org/pdf/2606.27828","project":null,"code":"https://github.com/Mrakas/video-mme-logical","data":null,"hfPaper":"https://huggingface.co/papers/2606.27828"},"evidence":{"snippet":"To isolate this capability, we introduce Video-MME-Logical, a controlled benchmark organized around five temporal-logical operations: state tracking, sequential counting, temporal ordering, dynamic spatiality, and structural composition.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":26,"hfDailySubmittedAt":"2026-06-30T00:00:00.000Z","githubStars":10,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.27828"},"ranking":{"90d":{"score":42,"rank":136,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates video temporal-logical reasoning in multimodal LLMs across 25 fine-grained task categories, with difficulty-controlled final-answer scoring and intermediate-state diagnostics.","whyItMatters":"Isolates temporal-logical capabilities from static recognition, revealing significant human-model gaps and providing a scalable testbed for analysis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c86aa692865b502b97e68aab95e557501c9e79f7922ecba42d53b13fb9dae09f"},"motivation":"Recent interest in multimodal large language models (MLLMs) raises a central question: can they reason over dynamic visual evidence rather than merely recognize objects or events in individual frames?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.27828","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Mrakas","organizationType":"academic-lab","sourceUrl":"https://github.com/Mrakas/video-mme-logical","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_videofdb_d27be39c","familyId":"bmf_e0d2574209dd","name":"VideoFDB","oneLine":"VideoFDB is a benchmark presented for evaluating full-duplex audio-visual conversational agents. It includes 237 dyadic clips with 11 nonverbal conversational dynamics from real-world video calls, along with a taxonomy and rubric-based LM-as-judge evaluation framework.","area":"Multimodal","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30256","pdf":"https://arxiv.org/pdf/2605.30256","project":"https://research.nvidia.com/labs/amri/projects/video-fdb/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30256"},"evidence":{"snippet":"In this work, we present VideoFDB, the first benchmark to evaluate full-duplex audio-visual-to-audio-visual (AV2AV) conversational agents.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30256"},"ranking":{},"description":"VideoFDB is a benchmark presented for evaluating full-duplex audio-visual conversational agents. It includes 237 dyadic clips with 11 nonverbal conversational dynamics from real-world video calls, along with a taxonomy and rubric-based LM-as-judge evaluation framework.","whyItMatters":"Existing full-duplex benchmarks only evaluate speech, missing the audio-visual nature of natural conversation. VideoFDB aims to fill this gap by evaluating agents that must produce and interpret nonverbal cues.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0d5639bb80e973a363a731f3c7ef71dec4880a7a85703f699fcfe88472a4ddf0"},"motivation":"Natural human conversation is full-duplex and audio-visual: people simultaneously speak and listen while continuously interpreting and producing nonverbal cues, such as nods, smiles, and gestures.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30256","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_videogaia_c89526e7","familyId":"bmf_be79630d2651","name":"VideoGAIA","oneLine":"VideoGAIA evaluates agentic video understanding through multi-turn, tool-augmented interactions where models must iteratively perceive videos, invoke external tools, and integrate multimodal evidence. It contains 271 human-verified tasks across diverse real-world scenarios.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-12","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.14718","pdf":"https://arxiv.org/pdf/2608.14718","project":null,"code":"https://github.com/zfkarl/VideoGAIA","data":null,"hfPaper":"https://huggingface.co/papers/2608.14718"},"evidence":{"snippet":"Towards this end, we introduce VideoGAIA, an agentic video understanding benchmark for general artificial intelligence (AI) assistants.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":10,"hfDailySubmittedAt":"2026-08-18T00:00:00.000Z","githubStars":16,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.14718"},"ranking":{"30d":{"score":43,"rank":44,"coverage":0.85,"confidence":"High"},"90d":{"score":43,"rank":133,"coverage":0.7,"confidence":"Medium"}},"description":"VideoGAIA evaluates agentic video understanding through multi-turn, tool-augmented interactions where models must iteratively perceive videos, invoke external tools, and integrate multimodal evidence. It contains 271 human-verified tasks across diverse real-world scenarios.","whyItMatters":"Conventional single-turn video understanding benchmarks are becoming saturated; VideoGAIA moves beyond to assess advanced MLLMs' ability to act as general AI assistants, using tools to gather complementary information across turns. It provides a timely, challenging benchmark for next-generation video understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"7d365d3edbf1067bad146b71b70188fdbdf07060e1d52b2a5e61a2de2f371938"},"motivation":"Video understanding is a fundamental task for evaluating the capabilities of multimodal large language models (MLLMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.14718","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Karl28","organizationType":"community","sourceUrl":"https://github.com/zfkarl/VideoGAIA","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_281510454193b470","familyId":"catalog_family_281510454193b470","name":"VideoHolmes","oneLine":"VideoHolmes evaluates video understanding and reasoning capabilities in multimodal models.","description":"VideoHolmes evaluates video understanding and reasoning capabilities in multimodal models.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Video"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/videoholmes","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_281510454193b470"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/videoholmes"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"videoholmes","url":"https://llm-stats.com/benchmarks/videoholmes","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","video"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_4dd9ff92b3fe2f97","familyId":"catalog_family_4dd9ff92b3fe2f97","name":"VideoMME w sub.","oneLine":"The first-ever comprehensive evaluation benchmark of Multi-modal LLMs in Video analysis. Features 900 videos (254 hours) with 2,700 question-answer pairs covering 6 primary visual domains and 30 subfields. Evaluates temporal understanding across short (11 seconds) to long (1 hour) videos with multi-modal inputs including video frames, subtitles, and audio.","description":"The first-ever comprehensive evaluation benchmark of Multi-modal LLMs in Video analysis. Features 900 videos (254 hours) with 2,700 question-answer pairs covering 6 primary visual domains and 30 subfields. Evaluates temporal understanding across short (11 seconds) to long (1 hour) videos with multi-modal inputs including video frames, subtitles, and audio.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/videomme-w-sub.","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4dd9ff92b3fe2f97"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/videomme-w-sub."}],"catalogSources":[{"catalog":"llm-stats","sourceId":"videomme-w-sub.","url":"https://llm-stats.com/benchmarks/videomme-w-sub.","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","video","vision"],"catalogModelCount":10,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_c5faee41a30faa01","familyId":"catalog_family_c5faee41a30faa01","name":"VideoMME w/o sub.","oneLine":"Video-MME is a comprehensive evaluation benchmark for multi-modal large language models in video analysis. It features 900 videos across 6 primary visual domains with 30 subfields, ranging from 11 seconds to 1 hour in duration, with 2,700 question-answer pairs. The benchmark evaluates MLLMs' capabilities in processing sequential visual data and multi-modal content including video frames, subtitles, and audio.","description":"Video-MME is a comprehensive evaluation benchmark for multi-modal large language models in video analysis. It features 900 videos across 6 primary visual domains with 30 subfields, ranging from 11 seconds to 1 hour in duration, with 2,700 question-answer pairs. The benchmark evaluates MLLMs' capabilities in processing sequential visual data and multi-modal content including video frames, subtitles, and audio.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/videomme-w-o-sub.","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c5faee41a30faa01"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/videomme-w-o-sub."}],"catalogSources":[{"catalog":"llm-stats","sourceId":"videomme-w-o-sub.","url":"https://llm-stats.com/benchmarks/videomme-w-o-sub.","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","video","vision"],"catalogModelCount":10,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_3f084ff3bbed7170","familyId":"catalog_family_3f084ff3bbed7170","name":"VideoMMMU","oneLine":"Video-MMMU evaluates Large Multimodal Models' ability to acquire knowledge from expert-level professional videos across six disciplines through three cognitive stages: perception, comprehension, and adaptation. Contains 300 videos and 900 human-annotated questions spanning Art, Business, Science, Medicine, Humanities, and Engineering.","description":"Video-MMMU evaluates Large Multimodal Models' ability to acquire knowledge from expert-level professional videos across six disciplines through three cognitive stages: perception, comprehension, and adaptation. Contains 300 videos and 900 human-annotated questions spanning Art, Business, Science, Medicine, Humanities, and Engineering.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Reasoning","Healthcare","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3f084ff3bbed7170"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/videommmu"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/videommmu"}],"catalogSources":[{"catalog":"benchlm","sourceId":"videoMmmu","url":"https://benchlm.ai/benchmarks/videommmu","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"VideoMMMU","format":"Video + text reasoning","tasks":"Video-grounded expert reasoning","successorKey":null},{"catalog":"llm-stats","sourceId":"videommmu","url":"https://llm-stats.com/benchmarks/videommmu","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","healthcare","vision"],"catalogModelCount":26,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_videoodyssey_004b7012","familyId":"bmf_62f3a2834934","name":"VideoOdyssey","oneLine":"VideoOdyssey evaluates models on ultra-long-context video understanding using videos averaging 109 minutes across 11 domains, with two subsets for visual and audio-visual understanding. Tasks include question answering with continuous certificates averaging 16 and 12.8 minutes respectively.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Long Context"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.22907","pdf":"https://arxiv.org/pdf/2605.22907","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.22907"},"evidence":{"snippet":"Driven by this metric, we introduce VideoOdyssey, a benchmark specifically designed for ultra-long-context and omni-modal video understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":4,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.22907"},"ranking":{},"description":"VideoOdyssey evaluates models on ultra-long-context video understanding using videos averaging 109 minutes across 11 domains, with two subsets for visual and audio-visual understanding. Tasks include question answering with continuous certificates averaging 16 and 12.8 minutes respectively.","whyItMatters":"Existing long-video benchmarks often only test short segments, failing to capture the cognitive load of continuous reasoning over long spans. VideoOdyssey's multi-level continuous certificates provide a diagnostic for evaluating model performance across varying context lengths, addressing a gap in measuring true long-context and omni-modal understanding.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"97f514bf9efdb638edb1da47ed2b9d10b6559c39c6935a43de73a20e82a9d00a"},"motivation":"Real-world long video understanding requires models to perform continuous tracking, information integration and memory retention over massive temporal spans within extreme video durations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.22907","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Long Context & Memory"],"domainScope":"general"},{"id":"bm_videorover-bench_f87e9ddf","familyId":"bmf_f0af8d10d7cf","name":"VideoRover-Bench","oneLine":"VideoRover-Bench is a benchmark for open-world video reasoning that combines video understanding with deep research, stratified by video duration and research difficulty. It evaluates the capability of models to locate sparse visual evidence and acquire external knowledge through coordinated video cropping, multimodal search, and webpage browsing.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":0.65,"links":{"report":"http://arxiv.org/abs/2608.23329v1","pdf":"https://arxiv.org/pdf/2608.23329v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We also introduce VideoRover-Bench, a benchmark stratified by video duration and research difficulty.","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23329"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"VideoRover-Bench is a benchmark for open-world video reasoning that combines video understanding with deep research, stratified by video duration and research difficulty. It evaluates the capability of models to locate sparse visual evidence and acquire external knowledge through coordinated video cropping, multimodal search, and webpage browsing.","whyItMatters":"The benchmark fills a gap by assessing unified video reasoning and multi-step information seeking, which are typically developed in isolation. It provides a structured evaluation to guide the development of video agents that can handle complex open-world tasks requiring external knowledge.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"b945bd507dc1d7358272622c9e1fb672358f5aa810b507c2a1892c63f8da16aa"},"motivation":"Open-world video understanding often requires a model to locate sparse visual evidence and acquire external knowledge that is absent from the video and its parametric memory.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with public evaluation protocol used to compare models and tool configurations.","canonicalNameSource":"abstract","canonicalNameEvidence":"We also introduce VideoRover-Bench, a benchmark stratified by video duration and research difficulty."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.23329v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":55,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses trending topics in video understanding and agentic AI, with clear release and evaluation framework."},"evaluationMode":"score_submission","publishers":[{"name":"VideoRover Team","organizationType":"academic-lab","sourceUrl":"http://arxiv.org/abs/2608.23329v1","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_8d535515b5577d52","familyId":"catalog_family_8d535515b5577d52","name":"VideoSimpleQA","oneLine":"VideoSimpleQA evaluates factual knowledge grounded in video content, measuring how accurately models answer short, fact-seeking questions about videos.","description":"VideoSimpleQA evaluates factual knowledge grounded in video content, measuring how accurately models answer short, fact-seeking questions about videos.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Knowledge","Multimodal","Video","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/videosimpleqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_8d535515b5577d52"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/videosimpleqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"videosimpleqa","url":"https://llm-stats.com/benchmarks/videosimpleqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","multimodal","video","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_videovibe_8a828619","familyId":"bmf_58daaf002b88","name":"VideoVIBE","oneLine":"Evaluates video-grounded diagnostic understanding of one-shot website generation. Approximately 1.7K Video QA instances from 6,338 verified failures across semantic-logical, visual-motion, structural-temporal, and functional categories. Scoring is based on model accuracy in multi-agent system V2Lens compared against baseline Video MLLMs.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.09573","pdf":"https://arxiv.org/pdf/2608.09573","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.09573"},"evidence":{"snippet":"We introduce VideoVIBE, a video-grounded benchmark that transforms human-operated webpage recordings into fine-grained diagnostic tasks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.09573"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates video-grounded diagnostic understanding of one-shot website generation. Approximately 1.7K Video QA instances from 6,338 verified failures across semantic-logical, visual-motion, structural-temporal, and functional categories. Scoring is based on model accuracy in multi-agent system V2Lens compared against baseline Video MLLMs.","whyItMatters":"Existing benchmarks often score isolated artifacts or final outcomes, lacking diagnostic insight. This benchmark provides a repeatable protocol to assess failure modes in generated webpages, enabling targeted model improvements for interactive website generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"d727bd676b33f4f6bb5b9b1547b0e4dad2f8f56b84ca9144f2ef0d368c35acd7"},"motivation":"Natural-language-driven \"vibe coding\" enables the one-shot generation of visually rich and interactive web applications, yet reliable assessment of their quality has not kept pace.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.09573","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_videoweaver_b86f354a","familyId":"bmf_339e1bf0a622","name":"VideoWeaver","oneLine":"VideoWeaver is an agent harness and benchmark for long video generation, with 16 task categories and 285 cases, evaluating agents via evidence-grounded judge on process and output.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.08091","pdf":"https://arxiv.org/pdf/2606.08091","project":null,"code":"https://github.com/JianhuiWei7/VideoWeaver","data":null,"hfPaper":"https://huggingface.co/papers/2606.08091"},"evidence":{"snippet":"We introduce VideoWeaver, an agent harness and benchmark that evaluates and evolves skills for long video generation, where an agent turns a single instruction into a long video by composing foundation skills into its own workflow rather than following a predefined pipeline.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":33,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08091"},"ranking":{"90d":{"score":49,"rank":72,"coverage":0.55,"confidence":"Low"}},"description":"VideoWeaver is an agent harness and benchmark for long video generation, with 16 task categories and 285 cases, evaluating agents via evidence-grounded judge on process and output.","whyItMatters":"General-purpose agents are underevaluated on long-horizon multimodal tasks; VideoWeaver offers a reproducible benchmark to assess and evolve agent skills for video generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"fdbb1580023e736af8479d8f50e230407c22dc1e90a1dded8e41bb195f62d981"},"motivation":"Recent agent frameworks such as Claude Code, Codex, and OpenClaw are strong at tool use and orchestration, but whether they can handle long video generation, a long-horizon multimodal task, remains underexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08091","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"JianhuiWei7","organizationType":"community","sourceUrl":"https://github.com/JianhuiWei7/VideoWeaver","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vidmsg_63a1ae32","familyId":"bmf_d9127847225b","name":"VidMsg","oneLine":"VidMsg is a benchmark for implicit message inference in short videos, containing 400 clips across 9 topic areas with retrieval and multiple-choice QA tasks.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03635","pdf":"https://arxiv.org/pdf/2606.03635","project":"https://iyttor.github.io/VidMsg","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03635"},"evidence":{"snippet":"We introduce VidMsg, a benchmark for evaluating implicit message understanding in short, internet-native video clips.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03635"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"VidMsg is a benchmark for implicit message inference in short videos, containing 400 clips across 9 topic areas with retrieval and multiple-choice QA tasks.","whyItMatters":"Addresses the lack of benchmarks for pragmatic video understanding, enabling evaluation of models on implicit message inference beyond visible content.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"e8a873af9cbd6852dba4f1d931e55f15e3e2eae042d53e6db63374fbdae8e59b"},"motivation":"Understanding short online videos involves more than identifying visible objects and actions; video makers often include an underlying message or purpose in the clip.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03635","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vietfashion_31e76260","familyId":"bmf_039d4698ef43","name":"VietFashion","oneLine":"VietFashion is a benchmark for sketch-text composed image retrieval centered on the Ao Dai, a traditional Vietnamese garment. It includes 650 sketches expanded to over 21,000 photorealistic images with captions, and adopts a multi-target retrieval setting.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Information retrieval"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.13427","pdf":"https://arxiv.org/pdf/2606.13427","project":"https://hng0303.github.io/VietFashion","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.13427"},"evidence":{"snippet":"We introduce VietFashion, a new benchmark for sketch-text composed image retrieval centered on the Ao Dai, a traditional Vietnamese garment.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.13427"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"VietFashion is a benchmark for sketch-text composed image retrieval centered on the Ao Dai, a traditional Vietnamese garment. It includes 650 sketches expanded to over 21,000 photorealistic images with captions, and adopts a multi-target retrieval setting.","whyItMatters":"Cultural garments require fine-grained retrieval systems that understand subtle structural and symbolic details. VietFashion exposes gaps in modeling cultural semantics and multi-modal composition.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"69626db0485cc05a310a0b2006a18317ac7387c1bb1737f741d0f33b2324ebe4"},"motivation":"Cultural garments pose a unique challenge for visual retrieval systems, as their identity often depends on subtle structural and symbolic details that are poorly captured by standard AI models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.13427","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Search & Retrieval"],"domainScope":"general"},{"id":"bm_virobench_ca7b2c7e","familyId":"bmf_3d16a3edd82a","name":"ViroBench","oneLine":"ViroBench evaluates nucleotide foundation models on viral genomics tasks across two axes: biological understanding (taxonomy classification, host prediction) and latent biosecurity risk (genome modeling, CDS completion). It includes 18 tasks, 58,314 viral samples, multiple split strategies (genus-disjoint, temporal), and 11 metrics, with a unified evaluation protocol and leaderboard.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech"],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.25388","pdf":"https://arxiv.org/pdf/2605.25388","project":null,"code":"https://github.com/QIANJINYDX/ViroBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.25388"},"evidence":{"snippet":"To address this, we introduce ViroBench, the first comprehensive and large-scale benchmark specifically designed for NFMs in viral settings.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":12,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25388"},"ranking":{},"description":"ViroBench evaluates nucleotide foundation models on viral genomics tasks across two axes: biological understanding (taxonomy classification, host prediction) and latent biosecurity risk (genome modeling, CDS completion). It includes 18 tasks, 58,314 viral samples, multiple split strategies (genus-disjoint, temporal), and 11 metrics, with a unified evaluation protocol and leaderboard.","whyItMatters":"Viral genomics lacks a unified evaluation standard for foundation models. ViroBench provides a diagnostic benchmark that measures both task performance and biosecurity risk, enabling model comparison and guiding development of safer, more robust viral nucleotide models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"922d8e86d3c93b75dd8274b046858b8510232f8630b545b8a53029625bb93f02"},"motivation":"Nucleotide sequences constitute the fundamental genetic basis of biological systems, rendering viral genomic analysis critical for biomedical advancement.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25388","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"QIANJINYDX / ViroBench team","organizationType":"community","sourceUrl":"https://github.com/QIANJINYDX/ViroBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_c00b13289041b12b","familyId":"catalog_family_c00b13289041b12b","name":"Virology Capabilities Test","oneLine":"Virology Capabilities Test (VCT) is an expert-level multiple-choice benchmark measuring the capability to troubleshoot complex virology laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.","description":"Virology Capabilities Test (VCT) is an expert-level multiple-choice benchmark measuring the capability to troubleshoot complex virology laboratory protocols. It evaluates dual-use biological knowledge relevant to bioweapons development.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Safety","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vct","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c00b13289041b12b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vct"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vct","url":"https://llm-stats.com/benchmarks/vct","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety","healthcare"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"bm_visanombench_44ae350c","familyId":"bmf_428dd43e7f91","name":"VisAnomBench","oneLine":"VisAnomBench is a benchmark assembled from public time-series datasets for anomaly detection, augmented with natural-language explanations selected from large vision-language models. It supports fine-tuning a parameter-efficient VLM called VisAnomReasoner for grounded anomaly detection decisions.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30344","pdf":"https://arxiv.org/pdf/2605.30344","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2605.30344"},"evidence":{"snippet":"To address this gap, we construct VisAnomBench, a curated benchmark built from public time-series datasets and augmented with high-quality anomaly explanations selected from multiple large VLMs using fine-grained, task-specific rewards.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30344"},"ranking":{},"description":"VisAnomBench is a benchmark assembled from public time-series datasets for anomaly detection, augmented with natural-language explanations selected from large vision-language models. It supports fine-tuning a parameter-efficient VLM called VisAnomReasoner for grounded anomaly detection decisions.","whyItMatters":"Public anomaly detection benchmarks typically lack natural-language rationales, hindering fine-tuning of VLMs for interpretable decisions. VisAnomBench addresses this gap by providing labeled explanations.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0e4f910773f25e58b960c61db8ef33d37831365e884ba0af93fa3ce81d50d057"},"motivation":"Recent advances in Vision-Language Models (VLMs) have achieved impressive performance across many tasks, yet prior studies report unsatisfactory performance when applying large language or multimodal models to finding abnormal patterns in sequential data.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30344","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_viseditbench_e7b0411b","familyId":"bmf_e2e3088b5cf4","name":"VisEditBench","oneLine":"VisEditBench is a benchmark comprising 1,395 human-annotated visualization code-editing tasks across two settings: feedback-guided repair and reference-guided restyling. Models are evaluated on their ability to revise existing visualization code based on multimodal feedback such as buggy or marked charts with textual instructions, and target chart images. Scoring is based on pass rates of generated code executions.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.10408","pdf":"https://arxiv.org/pdf/2608.10408","project":null,"code":"https://github.com/vis-nlp/VisEditBench","data":null,"hfPaper":"https://huggingface.co/papers/2608.10408"},"evidence":{"snippet":"We introduce VisEditBench, a benchmark of 1,395 human-annotated visualization code-editing tasks grounded in realistic visualization workflows and failure cases.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10408"},"ranking":{"30d":{"score":31,"rank":75,"coverage":0.55,"confidence":"Low"},"90d":{"score":31,"rank":230,"coverage":0.55,"confidence":"Low"}},"description":"VisEditBench is a benchmark comprising 1,395 human-annotated visualization code-editing tasks across two settings: feedback-guided repair and reference-guided restyling. Models are evaluated on their ability to revise existing visualization code based on multimodal feedback such as buggy or marked charts with textual instructions, and target chart images. Scoring is based on pass rates of generated code executions.","whyItMatters":"Existing benchmarks focus on generating visualizations from scratch, leaving the iterative editing process unexplored. VisEditBench provides a standardized evaluation for this practical task, enabling comparisons across VLMs and highlighting gaps in open-source models, particularly in visually grounded style adaptation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0140f504284b616dcae006538da4045f8c23ddf98da04ed0a9a58224967c4c7e"},"motivation":"Vision-language models (VLMs) have shown strong capabilities in generating visualization code from textual or visual specifications.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10408","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"vis-nlp","organizationType":"academic-lab","sourceUrl":"https://github.com/vis-nlp/VisEditBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_ef79cc168e97f387","familyId":"catalog_family_ef79cc168e97f387","name":"VisFactor","oneLine":"VisFactor is a benchmark evaluating fine-grained visual factor perception and reasoning over images.","description":"VisFactor is a benchmark evaluating fine-grained visual factor perception and reasoning over images.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/visfactor","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_ef79cc168e97f387"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/visfactor"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"visfactor","url":"https://llm-stats.com/benchmarks/visfactor","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_71bd6f729f139564","familyId":"catalog_family_71bd6f729f139564","name":"Vision2Web","oneLine":"Vision2Web evaluates multimodal models on converting visual designs and screenshots into functional web pages, measuring end-to-end design-to-code capability.","description":"Vision2Web evaluates multimodal models on converting visual designs and screenshots into functional web pages, measuring end-to-end design-to-code capability.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Code","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_71bd6f729f139564"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/vision2web"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vision2web"}],"catalogSources":[{"catalog":"benchlm","sourceId":"vision2Web","url":"https://benchlm.ai/benchmarks/vision2web","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"Vision2Web","format":"Visual reference to web implementation","tasks":"Screenshot-to-web tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"vision2web","url":"https://llm-stats.com/benchmarks/vision2web","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","code","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_vistahop_3279a79d","familyId":"bmf_acaea9923b4f","name":"VistaHop","oneLine":"VistaHop is a benchmark for long-horizon Visual DeepSearch, evaluating repeated image inspection, visual-anchor grounding, and evidence traversal across 600 tasks.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.03273","pdf":"https://arxiv.org/pdf/2606.03273","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.03273"},"evidence":{"snippet":"In this work, we introduce VistaHop, a benchmark designed specifically to evaluate Visual DeepSearch.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.03273"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"VistaHop is a benchmark for long-horizon Visual DeepSearch, evaluating repeated image inspection, visual-anchor grounding, and evidence traversal across 600 tasks.","whyItMatters":"Targets the gap in evaluating MLLMs' ability to iteratively revisit visual evidence and reason across multiple steps in complex visual queries.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"180afcc020a05bbbda354309b8734010508c08b97e785fdf8964636cd10b2463"},"motivation":"Visual DeepSearch tasks require multimodal large language models (MLLMs) to resolve complex visual queries by repeatedly inspecting image regions, grounding reasoning in visual evidence, and connecting fine-grained clues across multiple steps.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.03273","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vistarget-bench_aaa3c19c","familyId":"bmf_a8b611d92a57","name":"VisTarget-Bench","oneLine":"A 150-task human-verified benchmark pairing questions with held-out target images to separate image-retrieval failures from visual-perception failures in multimodal search agents.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Planning","Information retrieval"],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.28062","pdf":"https://arxiv.org/pdf/2608.28062","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"During post-training, our Failure-Aware GSPO (FA-GSPO) recovers salvageable abnormal rollouts and filters invalid ones to improve bounded multimodal planning and search.We also introduce VisTarget-Bench, a 150-task human-verified benchmark that pairs each question with a held-out target image, distinguishing image-retrieval failures from visual-perception failures.","reasonCodes":["exact named benchmark artifact released in abstract","evaluation protocol evidence","named entity is a model, method, framework, or dataset used in evaluation"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.28062"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A 150-task human-verified benchmark pairing questions with held-out target images to separate image-retrieval failures from visual-perception failures in multimodal search agents.","whyItMatters":"Targets a gap in evaluating visual grounding within search agent trajectories, enabling finer diagnosis of retrieval versus perception errors.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T04:51:30.440525Z","inputHash":"9f129758288a83acfa1cc6382fcc0477fedcd0fc4a2e3a2f4c2f2340cef950d4"},"motivation":"Multimodal search agents extend parametric knowledge with newly emerging and long-tail evidence from the open web.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T04:51:30.440525Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally introduced with a clear task definition and evaluation protocol, and code/tasks are reported as available in a public repository.","canonicalNameSource":"abstract","canonicalNameEvidence":"We also introduce VisTarget-Bench, a 150-task human-verified benchmark that pairs each question with a held-out target image, distinguishing image-retrieval failures from visual-perception failures."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.28062","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T01:03:30.163531Z"},"attentionForecast":{"score":65,"confidence":"Medium","horizon":"7d","reason":"The benchmark addresses a novel multimodal search evaluation gap and accompanies a system paper, attracting interest from agentic AI researchers."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception","Search & Retrieval"],"domainScope":"general"},{"id":"bm_vistr-bench_5dcce2bf","familyId":"bmf_744954eb90a9","name":"ViSTR-Bench","oneLine":"An evaluation suite of 1,340 video QA pairs across 15 subtasks assessing MLLM qualitative reasoning in dynamic scenes, covering motion perception, spatial relations, outcome prediction, and physical dynamics.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-23","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20868","pdf":"https://arxiv.org/pdf/2607.20868","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20868"},"evidence":{"snippet":"In this paper, we introduce the Visual Spatial-Temporal Reasoning Benchmark (ViSTR-Bench), a novel evaluation suite designed to systematically assess whether MLLMs can perform qualitative reasoning from continuous visual cues in dynamic scenes.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20868"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"An evaluation suite of 1,340 video QA pairs across 15 subtasks assessing MLLM qualitative reasoning in dynamic scenes, covering motion perception, spatial relations, outcome prediction, and physical dynamics.","whyItMatters":"Current MLLMs lag behind humans in intuitive spatial-temporal reasoning; this probe highlights those gaps but lacks a standalone reusable benchmark.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ae48f6b0717c5be8d9fc7c12b59a668785bbf50b112720db48735b9adf1332fe"},"motivation":"Multimodal Large Language Models (MLLMs) have achieved remarkable success across diverse expert-level tasks, but they still struggle with fundamental abilities that humans naturally develop through continuous observation of the real world, such as spatial perception and dynamic reasoning.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20868","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_visualflip_556bf533","familyId":"bmf_eb3ea10a5081","name":"VisualFLIP","oneLine":"VisualFLIP evaluates multimodal LLMs on visual reasoning with 1,374 images in paired perturbation tasks. Each pair has a fixed question but minimally changed visual evidence so the answer flips. Scoring uses pair accuracy and Collapse Rate to test evidence dependence in capabilities like cardinality, attribute, spatial, and logic.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-05","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07872","pdf":"https://arxiv.org/pdf/2606.07872","project":"https://didizhu-judy.github.io/VisualFLIP/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07872"},"evidence":{"snippet":"We introduce VisualFLIP, a paired benchmark with 1,374 images arranged as same-question perturbation pairs across cardinality, attribute, spatial, and logic tasks.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07872"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"VisualFLIP evaluates multimodal LLMs on visual reasoning with 1,374 images in paired perturbation tasks. Each pair has a fixed question but minimally changed visual evidence so the answer flips. Scoring uses pair accuracy and Collapse Rate to test evidence dependence in capabilities like cardinality, attribute, spatial, and logic.","whyItMatters":"Accuracy alone can hide flawed reasoning. VisualFLIP exposes whether models truly rely on task-critical visual changes, distinguishing robust grounding from guesswork. This helps select models for high-stakes visual tasks.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f885a1b3364ce9ea25680e46db3dd46f70e2747bab2b7ef4b08044cddcf7878b"},"motivation":"When a multimodal large language model answers a visual reasoning question correctly, is the prediction actually supported by the task-critical visual evidence?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07872","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"VisualFLIP team","organizationType":"academic-lab","sourceUrl":"https://didizhu-judy.github.io/VisualFLIP/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_visualleakbench_6618ffcf","familyId":"bmf_233962516067","name":"VisualLeakBench","oneLine":"VisualLeakBench is a 500-image benchmark for evaluating action-boundary propagation failures in vision-language agents, with stratified subsets and oracle diagnostics.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Multimodal"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.07595","pdf":"https://arxiv.org/pdf/2606.07595","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.07595"},"evidence":{"snippet":"We present VisualLeakBench, a diversified 500-image benchmark spanning UI, chat, document, form, and dashboard scenes, and evaluate a stratified 100-image agent subset with four production VLM systems under two workflows: note capture and external handoff.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.07595"},"ranking":{},"description":"VisualLeakBench is a 500-image benchmark for evaluating action-boundary propagation failures in vision-language agents, with stratified subsets and oracle diagnostics.","whyItMatters":"Targets a specific safety failure mode in VLAs, but lacks clear public reuse path and scoring contract details.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5e01a3ea88a8b3cd98e1cfcb723ffdbdd4119cb6d556b049c93f9bcd5aaaa601"},"motivation":"Vision-language agents increasingly consume screenshots, documents, and user interfaces before writing to memory, sending messages, or invoking external tools.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.07595","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_e656cd850fb42bd5","familyId":"catalog_family_e656cd850fb42bd5","name":"VisualWebBench","oneLine":"A multimodal benchmark designed to assess the capabilities of multimodal large language models (MLLMs) across web page understanding and grounding tasks. Comprises 7 tasks (captioning, webpage QA, heading OCR, element OCR, element grounding, action prediction, and action grounding) with 1.5K human-curated instances from 139 real websites across 87 sub-domains.","description":"A multimodal benchmark designed to assess the capabilities of multimodal large language models (MLLMs) across web page understanding and grounding tasks. Comprises 7 tasks (captioning, webpage QA, heading OCR, element OCR, element grounding, action prediction, and action grounding) with 1.5K human-curated instances from 139 real websites across 87 sub-domains.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Frontend Development","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/visualwebbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e656cd850fb42bd5"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/visualwebbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"visualwebbench","url":"https://llm-stats.com/benchmarks/visualwebbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","frontend development","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_f1add9362ddfe9cd","familyId":"catalog_family_f1add9362ddfe9cd","name":"VisuLogic","oneLine":"VisuLogic evaluates logical reasoning capabilities in visual contexts.","description":"VisuLogic evaluates logical reasoning capabilities in visual contexts.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/visulogic","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f1add9362ddfe9cd"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/visulogic"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"visulogic","url":"https://llm-stats.com/benchmarks/visulogic","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vitabench_d7ecb50f","familyId":"bmf_2e1c2e7b149d","name":"VitaBench","oneLine":"VitaBench 2.0 evaluates personalized and proactive agent behavior in long-term, multi-session user interactions across food delivery, in-store consumption, and online travel domains. Tasks are per-user sequences of subtasks requiring agents to infer, utilize, and update user preferences from fragmented interaction history, with an extensible memory interface for controlled comparison across memory architectures.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-26","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.27141","pdf":"https://arxiv.org/pdf/2605.27141","project":null,"code":"https://github.com/meituan-longcat/vitabench-2.0","data":null,"hfPaper":"https://huggingface.co/papers/2605.27141"},"evidence":{"snippet":"To address this gap, we introduce VitaBench 2.0, a benchmark for evaluating personalized and proactive agent behavior in long-term user interactions.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":20,"hfDailySubmittedAt":null,"githubStars":63,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.27141"},"ranking":{},"description":"VitaBench 2.0 evaluates personalized and proactive agent behavior in long-term, multi-session user interactions across food delivery, in-store consumption, and online travel domains. Tasks are per-user sequences of subtasks requiring agents to infer, utilize, and update user preferences from fragmented interaction history, with an extensible memory interface for controlled comparison across memory architectures.","whyItMatters":"Existing agent benchmarks focus on reasoning and tool use, overlooking the challenges of inferring and leveraging user preferences over time. This benchmark isolates personalization and proactivity in long-horizon tasks, providing a measurement of practical readiness for life-serving applications where models must act on implicit and evolving user needs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9f68e150f3c43b78ef0ca220cd8ebd80f69ec08e04ed58d8987ee39ec55825b5"},"motivation":"Large language models (LLMs) have evolved into interactive agents that collaborate with users in real-world tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.27141","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Meituan","organizationType":"company-research-lab","sourceUrl":"https://github.com/meituan-longcat/vitabench-2.0","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"vitaBench","url":"https://benchlm.ai/benchmarks/vitabench","paperUrl":"https://vitabench.github.io/","year":"2025","fullName":"VITA-Bench","format":"End-to-end interactive agent evaluation","tasks":"Interactive consumer-service agent tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"vita-bench","url":"https://llm-stats.com/benchmarks/vita-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","agents"],"catalogModelCount":10,"catalogStarCount":0},{"id":"catalog_a7ef41637e381635","familyId":"catalog_family_a7ef41637e381635","name":"VLADBench","oneLine":"VLADBench is a vision-language autonomous-driving benchmark evaluating understanding of dynamic traffic scenes and participants.","description":"VLADBench is a vision-language autonomous-driving benchmark evaluating understanding of dynamic traffic scenes and participants.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vladbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a7ef41637e381635"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vladbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vladbench","url":"https://llm-stats.com/benchmarks/vladbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_bd6ed6044f5fa2e0","familyId":"catalog_family_bd6ed6044f5fa2e0","name":"VLMsAreBiased","oneLine":"VLMsAreBiased evaluates whether vision-language models rely on visual evidence or fall back on language priors when answering.","description":"VLMsAreBiased evaluates whether vision-language models rely on visual evidence or fall back on language priors when answering.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vlmsarebiased","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_bd6ed6044f5fa2e0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vlmsarebiased"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vlmsarebiased","url":"https://llm-stats.com/benchmarks/vlmsarebiased","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","vision"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_b5d612c1a1f2eede","familyId":"catalog_family_b5d612c1a1f2eede","name":"VLMsAreBlind","oneLine":"A vision-language benchmark that probes blind spots and brittle reasoning in multimodal models.","description":"A vision-language benchmark that probes blind spots and brittle reasoning in multimodal models.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vlmsareblind","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_b5d612c1a1f2eede"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vlmsareblind"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vlmsareblind","url":"https://llm-stats.com/benchmarks/vlmsareblind","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vnish-global-operator-routing-benchmark_1814945a","familyId":"bmf_17f6ffb52087","name":"VNISH GLOBAL Multilingual Operator Routing Benchmark","oneLine":"Tests multilingual AI assistant routing across 80 cases in ten locales, requiring correct owned-source routing, hard stops on missing evidence, and citation precision.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-22","firstSeenAt":"2026-08-24","recognitionConfidence":0.4,"links":{"report":"https://github.com/vnish-global/vnish-global-operator-routing-benchmark","pdf":null,"project":"https://vnish.global/ai/","code":"https://github.com/vnish-global/vnish-global-operator-routing-benchmark","data":null,"hfPaper":null},"evidence":{"snippet":"vnish-global-operator-routing-benchmark Multilingual routing and safety benchmark for AI assistants using VNISH GLOBAL owned sources across vnish.global, vnish.ninja and roiasic.com.","reasonCodes":["discovered via github","benchmark term in abstract","evaluation protocol evidence","public artifact URL","no explicit benchmark release evidence"]},"dataStatus":"primary-source-candidate","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"github","id":"github:vnish-global/vnish-global-operator-routing-benchmark"},"ranking":{"30d":{"score":23,"rank":116,"coverage":0.55,"confidence":"Low"},"90d":{"score":23,"rank":320,"coverage":0.55,"confidence":"Low"}},"description":"Tests multilingual AI assistant routing across 80 cases in ten locales, requiring correct owned-source routing, hard stops on missing evidence, and citation precision.","whyItMatters":"It provides a focused, brand-owned evaluation for routing and safety properties that general agent benchmarks rarely measure.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:18:56.318436Z","inputHash":"ad81fecd0ead1f99f0fd50bd35ad71074830115accd4f70ae3126663608b9940"},"motivation":"vnish-global-operator-routing-benchmark Multilingual routing and safety benchmark for AI assistants using VNISH GLOBAL owned sources across vnish.global, vnish.ninja and roiasic.com.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The formal benchmark name is source-grounded, but the public release does not yet meet the independent evidence or adoption threshold."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://github.com/vnish-global/vnish-global-operator-routing-benchmark","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-24T07:04:32.475961Z"},"attentionForecast":{"score":38,"confidence":"Low","horizon":"7d","reason":"A small brand-specific routing benchmark is likely to draw limited early attention outside its targeted use case."},"evaluationMode":"public_reusable","publishers":[{"name":"VNISH GLOBAL","organizationType":"company-research-lab","sourceUrl":"https://vnish.global/ai/","role":"benchmark-publisher"}],"displayEligible":false,"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_5d8477303965671d","familyId":"catalog_family_5d8477303965671d","name":"VocalSound","oneLine":"A dataset for improving human vocal sounds recognition, containing over 21,000 crowdsourced recordings of laughter, sighs, coughs, throat clearing, sneezes, and sniffs from 3,365 unique subjects. Used for audio event classification and recognition of human non-speech vocalizations.","description":"A dataset for improving human vocal sounds recognition, containing over 21,000 crowdsourced recordings of laughter, sighs, coughs, throat clearing, sneezes, and sniffs from 3,365 unique subjects. Used for audio event classification and recognition of human non-speech vocalizations.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Audio"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vocalsound","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5d8477303965671d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vocalsound"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vocalsound","url":"https://llm-stats.com/benchmarks/vocalsound","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["audio"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_fb9ba53aee44d463","familyId":"catalog_family_fb9ba53aee44d463","name":"VoiceBench Avg","oneLine":"VoiceBench is the first benchmark designed to provide a multi-faceted evaluation of LLM-based voice assistants, evaluating capabilities including general knowledge, instruction-following, reasoning, and safety using both synthetic and real spoken instruction data with diverse speaker characteristics and environmental conditions.","description":"VoiceBench is the first benchmark designed to provide a multi-faceted evaluation of LLM-based voice assistants, evaluating capabilities including general knowledge, instruction-following, reasoning, and safety using both synthetic and real spoken instruction data with diverse speaker characteristics and environmental conditions.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Safety","Speech To Text","General","Communication"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/voicebench-avg","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fb9ba53aee44d463"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/voicebench-avg"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"voicebench-avg","url":"https://llm-stats.com/benchmarks/voicebench-avg","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","safety","speech to text","general","communication"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_65871f8ee38756ee","familyId":"catalog_family_65871f8ee38756ee","name":"VoxelBench Image","oneLine":"A live human-preference benchmark where multimodal models build voxel structures from image references and voters compare anonymous results produced from the same prompt.","description":"A live human-preference benchmark where multimodal models build voxel structures from image references and voters compare anonymous results produced from the same prompt.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://voxelbench.ai/leaderboard","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_65871f8ee38756ee"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/voxelbench-image"}],"catalogSources":[{"catalog":"benchlm","sourceId":"voxelbench-image","url":"https://benchlm.ai/benchmarks/voxelbench-image","paperUrl":"https://voxelbench.ai/leaderboard","year":"2025","fullName":"VoxelBench Image-Prompt Leaderboard","format":"Glicko-2 rating from blind pairwise votes","tasks":"Live image-reference prompts for 3D voxel construction","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_87636d519acb171b","familyId":"catalog_family_87636d519acb171b","name":"VoxelBench Text","oneLine":"A live human-preference benchmark where language models turn text prompts into voxel structures and voters compare anonymous builds from the same prompt.","description":"A live human-preference benchmark where language models turn text prompts into voxel structures and voters compare anonymous builds from the same prompt.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://voxelbench.ai/leaderboard","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_87636d519acb171b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/voxelbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"voxelbench","url":"https://benchlm.ai/benchmarks/voxelbench","paperUrl":"https://voxelbench.ai/leaderboard","year":"2025","fullName":"VoxelBench Text-Prompt Leaderboard","format":"Glicko-2 rating from blind pairwise votes","tasks":"Live text prompts for 3D voxel construction","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_2f2131e6cb30568d","familyId":"catalog_family_2f2131e6cb30568d","name":"VoxPopuli WER","oneLine":"A speech-recognition benchmark on the cleaned Artificial Analysis VoxPopuli subset, reported as word error rate where lower is better.","description":"A speech-recognition benchmark on the cleaned Artificial Analysis VoxPopuli subset, reported as word error rate where lower is better.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/datasets/ArtificialAnalysis/VoxPopuli-Cleaned-AA","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_2f2131e6cb30568d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/voxpopuliwer"}],"catalogSources":[{"catalog":"benchlm","sourceId":"voxPopuliWer","url":"https://benchlm.ai/benchmarks/voxpopuliwer","paperUrl":"https://huggingface.co/datasets/ArtificialAnalysis/VoxPopuli-Cleaned-AA","year":"2026","fullName":"VoxPopuli-Cleaned-AA Word Error Rate","format":"Word error rate","tasks":"Speech-to-text transcription","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_voxsumm_b7118a8e","familyId":"bmf_d932f4e6d176","name":"VoxSumm","oneLine":"VoxSumm evaluates joint summarization and translation of long-form spoken news. It comprises 10,045 BBC article-summary pairs across 24 languages and about 703 hours of speech, with scoring based on summarization and translation quality.","area":"Speech & Audio","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.SD"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.10359","pdf":"https://arxiv.org/pdf/2608.10359","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.10359"},"evidence":{"snippet":"We additionally introduce VoxSumm, the first multilingual and cross-lingual benchmark for this task, comprising 10,045 BBC article-summary pairs across 24 languages and encompassing approximately 703 hours of speech data.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10359"},"ranking":{"30d":{"score":35,"rank":null,"coverage":0.3,"confidence":"Low"},"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"VoxSumm evaluates joint summarization and translation of long-form spoken news. It comprises 10,045 BBC article-summary pairs across 24 languages and about 703 hours of speech, with scoring based on summarization and translation quality.","whyItMatters":"Existing benchmarks treat long-document summarization and speech translation separately, leaving a gap for cross-lingual summarization of spoken content. VoxSumm provides a publicly inspectable resource for developing systems that compress and translate long-form speech, aiding evaluation of multilingual instruction-following and cross-lingual generation.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"902cb002d670db7e66be50d926cf2e9e67139ff4e4bc8d07dece8efd01e1f473"},"motivation":"As information increasingly traverses linguistic boundaries, users require concise cross-lingual representations of long-form content.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10359","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"VoxSumm Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2608.10359","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_6c0328a0f8c820fb","familyId":"catalog_family_6c0328a0f8c820fb","name":"VQA-Rad","oneLine":"VQA-RAD (Visual Question Answering in Radiology) is the first manually constructed dataset of medical visual question answering containing 3,515 clinically generated visual questions and answers about radiology images. The dataset includes questions created by clinical trainees on 315 radiology images from MedPix covering head, chest, and abdominal scans, designed to support AI development for medical image analysis and improve patient care.","description":"VQA-RAD (Visual Question Answering in Radiology) is the first manually constructed dataset of medical visual question answering containing 3,515 clinically generated visual questions and answers about radiology images. The dataset includes questions created by clinical trainees on 315 radiology images from MedPix covering head, chest, and abdominal scans, designed to support AI development for medical image analysis and improve patient care.","area":"Multimodal","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Image To Text","Multimodal","Healthcare","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vqa-rad","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6c0328a0f8c820fb"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vqa-rad"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vqa-rad","url":"https://llm-stats.com/benchmarks/vqa-rad","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image to text","multimodal","healthcare","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_vqabench_22600a6b","familyId":"bmf_16fc6c20e67a","name":"VQABench","oneLine":"Evaluates 12 image preprocessing techniques for cloud VLM-based visual question answering across 3 VQA datasets and 4 commercial models, measuring accuracy, cost, and latency.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07861","pdf":"https://arxiv.org/pdf/2608.07861","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07861"},"evidence":{"snippet":"To fill this gap, we present VQABench, the first systematic benchmark that treats client-side input preprocessing as a controlled variable for cloud-VLM-based VQA.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07861"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Evaluates 12 image preprocessing techniques for cloud VLM-based visual question answering across 3 VQA datasets and 4 commercial models, measuring accuracy, cost, and latency.","whyItMatters":"Assesses the impact of client-side preprocessing on cost-quality trade-offs for offloaded VQA inference, offering practical guidance for system design.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"51bfe3bd7cb7dc9126d5da07383cad13b5e77a25a0fe6cdc4271a7b688d807cb"},"motivation":"Vision-language models (VLMs) are becoming a practical backend for mobile visual question answering (VQA) systems, enabling smartphones and smart glasses to answer users' questions about the physical world.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07861","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_d55072af4a2137af","familyId":"catalog_family_d55072af4a2137af","name":"VQAv2","oneLine":"VQAv2 is a balanced Visual Question Answering dataset that addresses language bias by providing complementary images for each question, forcing models to rely on visual understanding rather than language priors. It contains approximately twice the number of image-question pairs compared to the original VQA dataset.","description":"VQAv2 is a balanced Visual Question Answering dataset that addresses language bias by providing complementary images for each question, forcing models to rely on visual understanding rather than language priors. It contains approximately twice the number of image-question pairs compared to the original VQA dataset.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Image To Text","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vqav2","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d55072af4a2137af"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vqav2"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vqav2","url":"https://llm-stats.com/benchmarks/vqav2","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image to text","multimodal","reasoning","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_e7a89b60e2a2e710","familyId":"catalog_family_e7a89b60e2a2e710","name":"VQAv2 (test)","oneLine":"VQA v2.0 (Visual Question Answering v2.0) is a balanced dataset designed to counter language priors in visual question answering. It consists of complementary image pairs where the same question yields different answers, forcing models to rely on visual understanding rather than language bias. The dataset contains 1,105,904 questions across 204,721 COCO images, requiring understanding of vision, language, and commonsense knowledge.","description":"VQA v2.0 (Visual Question Answering v2.0) is a balanced dataset designed to counter language priors in visual question answering. It consists of complementary image pairs where the same question yields different answers, forcing models to rely on visual understanding rather than language bias. The dataset contains 1,105,904 questions across 204,721 COCO images, requiring understanding of vision, language, and commonsense knowledge.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Image To Text","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vqav2-(test)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e7a89b60e2a2e710"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vqav2-(test)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vqav2-(test)","url":"https://llm-stats.com/benchmarks/vqav2-(test)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image to text","multimodal","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_afc35271db09b185","familyId":"catalog_family_afc35271db09b185","name":"VQAv2 (val)","oneLine":"VQAv2 is a balanced Visual Question Answering dataset containing open-ended questions about images that require understanding of vision, language, and commonsense knowledge to answer. VQAv2 addresses bias issues from the original VQA dataset by collecting complementary images such that every question is associated with similar images that result in different answers, forcing models to actually understand visual content rather than relying on language priors.","description":"VQAv2 is a balanced Visual Question Answering dataset containing open-ended questions about images that require understanding of vision, language, and commonsense knowledge to answer. VQAv2 addresses bias issues from the original VQA dataset by collecting complementary images such that every question is associated with similar images that result in different answers, forcing models to actually understand visual content rather than relying on language priors.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Image To Text","Language","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/vqav2-(val)","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_afc35271db09b185"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/vqav2-(val)"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"vqav2-(val)","url":"https://llm-stats.com/benchmarks/vqav2-(val)","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["image to text","language","multimodal","reasoning","vision"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_vsro-200_f4255395","familyId":"bmf_76014c7782c1","name":"VSRo-200","oneLine":"A large-scale dataset for visual speech recognition in Romanian, with 200 hours of video and annotations. It studies supervision quality, robustness under domain shift, and multimodal fusion.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Robustness"],"topics":["Multimodal"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.08112","pdf":"https://arxiv.org/pdf/2607.08112","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.08112"},"evidence":{"snippet":"Building on this dataset, we establish a benchmark for visual speech recognition in low-resource settings.","reasonCodes":["exact coined title identity tied to benchmark evidence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.08112"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A large-scale dataset for visual speech recognition in Romanian, with 200 hours of video and annotations. It studies supervision quality, robustness under domain shift, and multimodal fusion.","whyItMatters":"The dataset enables research in low-resource visual speech recognition, but the paper primarily focuses on studying supervision and robustness rather than defining a fixed benchmark with a scoring contract. It is a dataset resource for training, not a standalone evaluation benchmark.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"8ba8ecc8b6641fbc945e1839ad3d4dd577d9fdde76d84ce9b6f52041124f65c3"},"motivation":"We introduce VSRo-200, the first large-scale dataset for visual speech recognition (lip reading) in Romanian, comprising 200 hours of real-world podcast videos.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.08112","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"bm_vtc-bench_2b0c0459","familyId":"bmf_cbcf9d8c29b3","name":"VTC-Bench","oneLine":"VTC-Bench evaluates multiple LLM generations across five domains using Validated Task Coverage (VTC), which measures the number of distinct useful results obtained within k attempts. Tasks are selected from real data where output quality and task-relevant distinctness can be checked automatically without model-based judges.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-25","firstSeenAt":"2026-08-26","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.24228","pdf":"https://arxiv.org/pdf/2608.24228","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce VTC-Bench, a five-domain benchmark for this setting, together with Validated Task Coverage (VTC) as its core evaluation quantity.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.24228"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"VTC-Bench evaluates multiple LLM generations across five domains using Validated Task Coverage (VTC), which measures the number of distinct useful results obtained within k attempts. Tasks are selected from real data where output quality and task-relevant distinctness can be checked automatically without model-based judges.","whyItMatters":"Traditional evaluations focus on individual outputs or reduce multiple samples to a single score, missing the value of diverse useful results. VTC-Bench provides a reproducible metric for candidate sets, revealing differences in model behavior that single-draw quality metrics do not capture, aiding in model selection for applications requiring multiple outputs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"94ed14c421c58b3b609e3c9264d72fc41e77c77d3796e0ff8a014c90ee56b468"},"motivation":"Many LLM applications are most useful when they provide several candidate outputs for comparison, validation, or combination.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-26T06:08:47.245811Z","model":"deepseek-v4-pro","decisionReason":"The paper formally introduces a named benchmark and evaluation metric, and the abstract indicates publicly reproducible tasks without model-based judges.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce VTC-Bench, a five-domain benchmark for this setting, together with Validated Task Coverage (VTC) as its core evaluation quantity."},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.24228","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-26T06:08:22.244942Z"},"attentionForecast":{"score":70,"confidence":"Medium","horizon":"7d","reason":"Broad relevance to LLM evaluation with a novel metric and public availability likely attract attention from AI researchers."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_vul4py_70948b69","familyId":"bmf_f19614b79970","name":"Vul4Py","oneLine":"Vul4Py evaluates automated vulnerability repair in Python across 100 real vulnerabilities from 60 open-source projects, with paired exploit and functional oracles.","area":"Code & Software","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00692","pdf":"https://arxiv.org/pdf/2608.00692","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00692"},"evidence":{"snippet":"We present Vul4Py, a Python AVR benchmark in which every entry carries a paired oracle: an exploit oracle that must fail on the vulnerable revision and pass on the fixed one, together with a project-native pytest functional oracle that must pass on both.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00692"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"Vul4Py evaluates automated vulnerability repair in Python across 100 real vulnerabilities from 60 open-source projects, with paired exploit and functional oracles.","whyItMatters":"It addresses the gap of missing functional regression checks in existing Python AVR benchmarks, providing a more reliable comparison of repair methods.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f35fb45d613ec02aec2c8635a49dd69e31c95fdcf1c2def8036527a936117c62"},"motivation":"Automated Vulnerability Repair (AVR) has advanced rapidly across program analysis, machine learning, and Large Language Models (LLMs), but a verifiable, head-to-head comparison of AVR approaches on Python is still missing.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00692","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Coding & Software Engineering"],"domainScope":"specific"},{"id":"bm_vulbench-cpp_84620b34","familyId":"bmf_12ecf4f9c98a","name":"VULBENCH-CPP","oneLine":"VULBENCH-CPP evaluates the safety of AI-generated C++ code using multi-tier verification including functional testing, static analysis, dynamic analysis, and bounded model checking. It includes 8,918 programs from three LLMs and human authors.","area":"Safety & Trustworthiness","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":["Code generation"],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.00107","pdf":"https://arxiv.org/pdf/2607.00107","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.00107"},"evidence":{"snippet":"We introduce VULBENCH-CPP, a benchmark of 8,918 C++ programs from three open-weight LLMs (Gemma 3 27B IT, LLaMA 3.3 70B Instruct, Qwen 2.5 Coder 32B Instruct) and human authors across 851 competitive-programming tasks.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.00107"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"VULBENCH-CPP evaluates the safety of AI-generated C++ code using multi-tier verification including functional testing, static analysis, dynamic analysis, and bounded model checking. It includes 8,918 programs from three LLMs and human authors.","whyItMatters":"Security of AI-generated code is critical, but evaluations often rely on a single method. VULBENCH-CPP provides a comprehensive multi-tier benchmark that reveals AI code is more prone to runtime violations and that no single verification tier is sufficient.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"261920b3c0aa5b0e431cd835e8c27561c77eadfbb2f834c9e0caaa2a72e39042"},"motivation":"As large language models (LLMs) are increasingly deployed for systems programming, their ability to generate secure C++ code, where a single memory-safety failure creates an exploitable vulnerability, remains a critical concern.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.00107","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Safety & Trustworthiness","Coding & Software Engineering"],"domainScope":"specific"},{"id":"catalog_df879e8ccbd6d387","familyId":"catalog_family_df879e8ccbd6d387","name":"VulcanBench CII v1","oneLine":"A post-cutoff software-engineering benchmark with hidden functional tests and regression guards, reported for vendor coding-agent harnesses.","description":"A post-cutoff software-engineering benchmark with hidden functional tests and regression guards, reported for vendor coding-agent harnesses.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://github.com/morganlinton/VulcanBench/blob/main/docs/results/cii-v1-2026-08/README.md","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_df879e8ccbd6d387"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/vulcanciiv1"}],"catalogSources":[{"catalog":"benchlm","sourceId":"vulcanCiiV1","url":"https://benchlm.ai/benchmarks/vulcanciiv1","paperUrl":"https://github.com/morganlinton/VulcanBench/blob/main/docs/results/cii-v1-2026-08/README.md","year":"2026","fullName":"VulcanBench Coding Intelligence Index v1","format":"Pass@1 with vendor coding-agent harnesses","tasks":"38 validated post-cutoff repository tasks","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_09a2daba9ae50516","familyId":"catalog_family_09a2daba9ae50516","name":"VulcanBench v3","oneLine":"An open software-engineering benchmark built from real merged post-cutoff pull requests across Python, Rust, TypeScript, JavaScript, and Go repositories.","description":"An open software-engineering benchmark built from real merged post-cutoff pull requests across Python, Rust, TypeScript, JavaScript, and Go repositories.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://github.com/morganlinton/VulcanBench/tree/main","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_09a2daba9ae50516"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/vulcanbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"vulcanBench","url":"https://benchlm.ai/benchmarks/vulcanbench","paperUrl":"https://github.com/morganlinton/VulcanBench/tree/main","year":"2026","fullName":"VulcanBench v3","format":"Pass@1 with low, medium, and high effort","tasks":"23 post-cutoff repository tasks in the v3 report","successorKey":null}],"catalogCategories":["coding"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_vvm-bench_539536ab","familyId":"bmf_639ec828a5aa","name":"VVM-Bench","oneLine":"VVM-Bench evaluates Large Multimodal Models on semantic perception and modality understanding across six real and synthetic modalities, using multiple-choice questions and generation tasks to assess zero-shot generalization to unseen visual modalities.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-07-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.10308","pdf":"https://arxiv.org/pdf/2607.10308","project":null,"code":"https://github.com/Hunter-Will/VVM-Tuning","data":null,"hfPaper":"https://huggingface.co/papers/2607.10308"},"evidence":{"snippet":"To facilitate research in this direction, we introduce VVM-Bench, a comprehensive benchmark featuring 6 real and synthetic modalities to evaluate semantic perception and modality understanding.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.10308"},"ranking":{"90d":{"score":15,"rank":401,"coverage":0.7,"confidence":"Medium"}},"description":"VVM-Bench evaluates Large Multimodal Models on semantic perception and modality understanding across six real and synthetic modalities, using multiple-choice questions and generation tasks to assess zero-shot generalization to unseen visual modalities.","whyItMatters":"VVM-Bench provides a standardized protocol for assessing LMMs' ability to generalize across visual modalities, enabling comparison of current models on a crucial capability for real-world deployment where sensor types vary.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c76694826111f1cfc36a1e719f1ef9a93dc423d53d11b6557795d97a790a8f7d"},"motivation":"Despite the advancements of Large Multimodal Models (LMMs) in RGB vision, their ability to generalize to unseen visual modalities remains a largely unexplored challenge.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"European Conference on Computer Vision (ECCV) 2026","evidence":"Accepted by the European Conference on Computer Vision (ECCV) 2026","evidenceUrl":"https://arxiv.org/abs/2607.10308","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"European Conference on Computer Vision (ECCV) 2026","reviewStatus":"accepted","decisionRaw":"Accepted by the European Conference on Computer Vision (ECCV) 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.10308","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted by the European Conference on Computer Vision (ECCV) 2026","level":"author-claim"}]}],"publishers":[{"name":"VVM-Tuning project","organizationType":"academic-lab","sourceUrl":"https://github.com/Hunter-Will/VVM-Tuning","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_dfab7e9348194010","familyId":"catalog_family_dfab7e9348194010","name":"VWT2k-lite","oneLine":"A lighter multilingual benchmark slice published in provider tables for broad cross-lingual transfer and understanding.","description":"A lighter multilingual benchmark slice published in provider tables for broad cross-lingual transfer and understanding.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multilingual"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_dfab7e9348194010"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/vwt2klite"}],"catalogSources":[{"catalog":"benchlm","sourceId":"vwt2kLite","url":"https://benchlm.ai/benchmarks/vwt2klite","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"VWT2k-lite","format":"Cross-lingual benchmark","tasks":"Multilingual transfer tasks","successorKey":null}],"catalogCategories":["multilingual"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_wade_93cc3c8f","familyId":"bmf_13f7eac6f16c","name":"WADE","oneLine":"A reasoning-annotated benchmark for floating waste detection and grounding, with 2,167 images, 13,608 bounding boxes, and ten categories, each with recognition rules.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.22950v1","pdf":"https://arxiv.org/pdf/2608.22950v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce WADE, a reasoning-annotated benchmark containing 2,167 images from rural Bangladesh, 13,608 bounding boxes, and ten waste categories.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22950"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A reasoning-annotated benchmark for floating waste detection and grounding, with 2,167 images, 13,608 bounding boxes, and ten categories, each with recognition rules.","whyItMatters":"Provides a challenging benchmark for dense multi-instance grounding in a real-world environmental monitoring context, evaluating compact VLMs under various settings.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"cdb29c064fae6836489dede3760d522e205da2ee9c7df13fe38dc25e7abbe9ed"},"motivation":"Floating waste in inland waterways threatens aquatic ecosystems and requires timely monitoring under cluttered, multi-object conditions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named dataset benchmark with clear evaluation metrics and public release.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce WADE, a reasoning-annotated benchmark containing 2,167 images"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22950v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":25,"confidence":"Low","horizon":"7d","reason":"Specialized environmental AI topic with limited topical breadth; no public artifacts to generate early interest."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_waspmot_953ea5b1","familyId":"bmf_c1b6a961c222","name":"WaspMOT","oneLine":"A benchmark for long-term multi-object tracking of Trichogramma wasps in controlled ecological experiments, with 10 sequences of ~12,000 frames each and dense annotations. It evaluates identity preservation over extended durations.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.08729","pdf":"https://arxiv.org/pdf/2607.08729","project":null,"code":"https://github.com/tstanczyk95/WaspMOT/","data":null,"hfPaper":"https://huggingface.co/papers/2607.08729"},"evidence":{"snippet":"We introduce WaspMOT, a benchmark designed to address this gap through long-duration tracking of Trichogramma wasps in controlled ecological experiments.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.08729"},"ranking":{"90d":{"score":23,"rank":373,"coverage":0.55,"confidence":"Low"}},"description":"A benchmark for long-term multi-object tracking of Trichogramma wasps in controlled ecological experiments, with 10 sequences of ~12,000 frames each and dense annotations. It evaluates identity preservation over extended durations.","whyItMatters":"Existing MOT benchmarks focus on short videos, which do not assess long-term identity preservation. This benchmark provides a controlled scenario with closed-set tracking, potentially revealing limitations in current methods that are not observable in conventional datasets.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"6741da454f366277ac0714d7dd5188deaace448bb17b8d19baa950289deb744e"},"motivation":"Multi-object tracking (MOT) has achieved strong performance on benchmarks dominated by short video sequences.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"publication_reported","venue":"AVSS 2026","evidence":"AVSS 2026","evidenceUrl":"https://arxiv.org/abs/2607.08729","source":"arxiv-journal-reference","evidenceLevel":"strong-author-metadata","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publications":[{"venueName":"AVSS 2026","publicationStatus":"published","evidence":[{"sourceType":"arxiv-journal-reference","sourceUrl":"https://arxiv.org/abs/2607.08729","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"AVSS 2026","level":"strong-author-metadata"}]}],"capabilityGroups":["Multimodal Perception","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_watchact_107de5f7","familyId":"bmf_84d3a4a518a6","name":"WatchAct","oneLine":"WatchAct evaluates robot manipulation from observed human behavior. Each instance pairs a human-action video and a language instruction with an aligned simulator scene and an executable LIBERO task. It covers 3,000 long-horizon instances across 14 tasks in four capability domains: Event Grounding, Procedural Reasoning, Implicit Intent Inference, and Episodic Reasoning. The evaluation protocol separately measures video-to-plan reasoning, policy execution under oracle plans, and full task completion, in simulation and on a Franka Research 3 robot.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Interactive Environment","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.26443","pdf":"https://arxiv.org/pdf/2606.26443","project":"https://baiqi-li.github.io/watchact_page/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.26443"},"evidence":{"snippet":"We introduce WatchAct, a benchmark for robot manipulation grounded in observed human behavior.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.26443"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"WatchAct evaluates robot manipulation from observed human behavior. Each instance pairs a human-action video and a language instruction with an aligned simulator scene and an executable LIBERO task. It covers 3,000 long-horizon instances across 14 tasks in four capability domains: Event Grounding, Procedural Reasoning, Implicit Intent Inference, and Episodic Reasoning. The evaluation protocol separately measures video-to-plan reasoning, policy execution under oracle plans, and full task completion, in simulation and on a Franka Research 3 robot.","whyItMatters":"Existing manipulation benchmarks typically evaluate from a single current image, lacking grounding in observed human behavior. WatchAct fills this gap by assessing robots' ability to reason about events, procedures, intents, and scene changes from video, which is critical for real-world human-robot collaboration. It provides a disentangled evaluation to isolate reasoning and execution failures, offering practical value for developing and comparing systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"19fa056d94a3b2468d06aa862d3835c64ee43c1fa5bb28ac89328e2ec606faa3"},"motivation":"A robot working alongside people must reason about what they have done, in what order, and with what intent.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.26443","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_waveformqa_f89d742c","familyId":"bmf_7ce4ac0aa81a","name":"WaveformQA","oneLine":"WaveformQA is a QA benchmark for LLM temporal reasoning over digital waveforms, comprising 360 questions with programmatically generated ground truths across eight categories, including multi-signal correlation and event ordering, with waveforms generated from open-source designs.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Paper only","releasedAt":"2026-07-22","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.20638","pdf":"https://arxiv.org/pdf/2607.20638","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.20638"},"evidence":{"snippet":"This paper presents WaveformQA, an open-source question-answering benchmark for evaluating LLM temporal reasoning over digital waveforms.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.20638"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"WaveformQA is a QA benchmark for LLM temporal reasoning over digital waveforms, comprising 360 questions with programmatically generated ground truths across eight categories, including multi-signal correlation and event ordering, with waveforms generated from open-source designs.","whyItMatters":"Temporal reasoning over waveforms is critical for hardware verification; existing benchmarks focus on HDL generation, leaving this capability untested. WaveformQA provides a reproducible way to evaluate LLMs on this task.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"83705100d08ccc14f829553cc7676c0203b024ca7d1b0a7a3f1f30470eaec649"},"motivation":"Large Language Models (LLMs) have demonstrated strong capabilities in code generation and reasoning, yet their ability to perform temporal reasoning over digital waveform data remains largely unexplored.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.20638","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"WaveformQA Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2607.20638","role":"benchmark-publisher"}],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_wbench_de4d593c","familyId":"bmf_99d5a953e34c","name":"WBench","oneLine":"WBench evaluates interactive video world models across five dimensions: video quality, setting adherence, interaction adherence, consistency, and physics compliance. It includes 289 test cases and 1,058 multi-turn interaction sequences, covering diverse scenes and control types, with 22 automatic sub-metrics validated against human judgment.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2605.25874","pdf":"https://arxiv.org/pdf/2605.25874","project":"https://meituan-longcat.github.io/WBench/","code":"https://github.com/meituan-longcat/WBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.25874"},"evidence":{"snippet":"To fill this gap, we introduce WBench, a comprehensive multi-turn benchmark for interactive world model evaluation along five dimensions, namely video quality, setting adherence, interaction adherence, consistency, and physics compliance.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":106,"hfDailySubmittedAt":null,"githubStars":226,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.25874"},"ranking":{},"description":"WBench evaluates interactive video world models across five dimensions: video quality, setting adherence, interaction adherence, consistency, and physics compliance. It includes 289 test cases and 1,058 multi-turn interaction sequences, covering diverse scenes and control types, with 22 automatic sub-metrics validated against human judgment.","whyItMatters":"No unified standard previously existed for evaluating interactive world models across the required competencies. WBench provides a comprehensive, multi-turn benchmark with a public leaderboard and open data/code, enabling systematic model comparison and diagnostic insights into strengths and weaknesses.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"31e9c387ff6cf694855b2bd000db561961b8980c31b4dac88fe1ec3f7dc74fa8"},"motivation":"Interactive world models are advancing rapidly, yet existing benchmarks cover only part of the required competencies, leaving no unified standard for systematic evaluation.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.25874","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Meituan LongCat Team","organizationType":"company-research-lab","sourceUrl":"https://github.com/meituan-longcat/WBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_02608907f64603da","familyId":"catalog_family_02608907f64603da","name":"We-Math","oneLine":"We-Math evaluates multimodal models on visual mathematical reasoning, requiring models to understand and solve math problems presented with visual elements such as diagrams, charts, and geometric figures.","description":"We-Math evaluates multimodal models on visual mathematical reasoning, requiring models to understand and solve math problems presented with visual elements such as diagrams, charts, and geometric figures.","area":"Mathematical Reasoning","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Math","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_02608907f64603da"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/wemath"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/we-math"}],"catalogSources":[{"catalog":"benchlm","sourceId":"weMath","url":"https://benchlm.ai/benchmarks/wemath","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"We-Math","format":"Multimodal mathematical reasoning","tasks":"Visually grounded math problems","successorKey":null},{"catalog":"llm-stats","sourceId":"we-math","url":"https://llm-stats.com/benchmarks/we-math","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","math","reasoning","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Mathematics & Formal Sciences"],"domainScope":"general"},{"id":"bm_weavebench_27cb863c","familyId":"bmf_88fdd16b4aa1","name":"WeaveBench","oneLine":"WeaveBench evaluates computer-use agents on 114 long-horizon tasks across 8 real-world work domains. Each task interleaves GUI interaction with command-line and code operations in a single trajectory, with scoring based on a trajectory-aware judge that detects fabricated evidence.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-08","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.09426","pdf":"https://arxiv.org/pdf/2606.09426","project":"https://weavebench.github.io/","code":"https://github.com/weavebench/WeaveBench","data":"https://huggingface.co/datasets/wanlilll/WeaveBench","hfPaper":"https://huggingface.co/papers/2606.09426"},"evidence":{"snippet":"Thus, we introduce WeaveBench, a long-horizon hybrid-interface benchmark with 114 tasks across 8 real-world work domains, grounded in real user requests and publicly verifiable artifacts.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":107,"hfDailySubmittedAt":"2026-06-12T00:00:00.000Z","githubStars":160,"githubScope":"benchmark_repo","hfDatasetDownloads":1573,"hfDatasetLikes":8},"source":{"type":"arxiv","id":"2606.09426"},"ranking":{"90d":{"score":69,"rank":8,"coverage":1.0,"confidence":"High","datasetDownloadRank":12,"datasetRankPopulation":66}},"description":"WeaveBench evaluates computer-use agents on 114 long-horizon tasks across 8 real-world work domains. Each task interleaves GUI interaction with command-line and code operations in a single trajectory, with scoring based on a trajectory-aware judge that detects fabricated evidence.","whyItMatters":"Existing benchmarks often evaluate interfaces separately, leaving hybrid orchestration under-tested. WeaveBench fills this gap by requiring agents to combine GUI and CLI/code in realistic tasks, and exposes that outcome-only grading overestimates performance.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"af8ec21e146fe5a6035ccd9fe211bba62061a42025bd35f8560232cc962ab2ae"},"motivation":"Computer-use agents (CUAs) increasingly operate in runtimes that combine visual desktop control, command-line execution, code editing, browsers, and external tools.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"source-reviewed","reviewedAt":"2026-08-30","sources":["https://arxiv.org/abs/2606.09426","https://weavebench.github.io/","https://github.com/weavebench/WeaveBench","https://huggingface.co/datasets/wanlilll/WeaveBench"]},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.09426","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"WeaveBench team","organizationType":"academic-lab","sourceUrl":"https://github.com/weavebench/WeaveBench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_013cc061eb74a965","familyId":"catalog_family_013cc061eb74a965","name":"Web Bench","oneLine":"Web Bench evaluates agents on realistic web-development engineering tasks, measuring end-to-end implementation in browser-based workflows.","description":"Web Bench evaluates agents on realistic web-development engineering tasks, measuring end-to-end implementation in browser-based workflows.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/web-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_013cc061eb74a965"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/web-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"web-bench","url":"https://llm-stats.com/benchmarks/web-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","coding"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_3e782f5c67e37a2b","familyId":"catalog_family_3e782f5c67e37a2b","name":"Web Search Index","oneLine":"A Vals AI comparison of native provider search and Exa across finance-analysis and legal-research tasks.","description":"A Vals AI comparison of native provider search and Exa across finance-analysis and legal-research tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.vals.ai/benchmarks/web_search","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3e782f5c67e37a2b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/valswebsearchindex"}],"catalogSources":[{"catalog":"benchlm","sourceId":"valsWebSearchIndex","url":"https://benchlm.ai/benchmarks/valswebsearchindex","paperUrl":"https://www.vals.ai/benchmarks/web_search","year":"2026","fullName":"Vals Web Search Index","format":"Accuracy by model and search-tool combination","tasks":"Finance Agent Benchmark v2 and Legal Research Benchmark tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"lib_webarena","familyId":"family_webarena","name":"WebArena","oneLine":"Established benchmark family · Computer Use.","area":"Computer Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Computer Use"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2023-01-01","releaseDatePrecision":"year","firstRelease":{"year":2023,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2307.13854","pdf":null,"project":"https://webarena.dev/","code":"https://github.com/web-arena-x/webarena","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_webarena"},"ranking":{},"recordType":"family","aliases":[],"sourceAttribution":[{"role":"official-project","url":"https://webarena.dev/"}],"adoptionRefs":["google-gemini25"],"modelReportReferences":[{"sourceId":"google-gemini25","url":"https://arxiv.org/abs/2507.06261","provider":"Google DeepMind","model":"Gemini 2.5"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Computer Use"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"webArena","url":"https://benchlm.ai/benchmarks/webarena","paperUrl":"https://arxiv.org/abs/2307.13854","year":"2024","fullName":"WebArena Web Agent Benchmark","format":"End-state task success","tasks":"812 long-horizon browser tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0},{"id":"lib_webarena_verified","familyId":"family_webarena","name":"WebArena-Verified","oneLine":"Established benchmark variant · Computer Use.","area":"Computer Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Computer Use"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2025-01-01","releaseDatePrecision":"year","firstRelease":{"year":2025,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://webarena.dev/","pdf":null,"project":"https://webarena.dev/","code":"https://github.com/web-arena-x/webarena","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_webarena_verified"},"ranking":{},"recordType":"variant","aliases":["WebArena Verified"],"sourceAttribution":[{"role":"official-project","url":"https://webarena.dev/"}],"adoptionRefs":["openai-gpt5"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"}],"catalogDiscoveryRefs":[],"catalogDiscoverySources":[],"usageObservations":[],"variantOf":"lib_webarena","capabilityGroups":["Computer Use"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"webArenaVerified","url":"https://benchlm.ai/benchmarks/webarena-verified","paperUrl":"https://openreview.net/forum?id=94tlGxmqkN","year":"2025","fullName":"WebArena-Verified Browser Agent Benchmark","format":"Deterministic end-state task success","tasks":"812 verified tasks; separate 258-task Hard subset","successorKey":null},{"catalog":"llm-stats","sourceId":"webarena-verified","url":"https://llm-stats.com/benchmarks/webarena-verified","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","multimodal","agents","vision"],"catalogModelCount":1,"catalogStarCount":0},{"id":"catalog_161771f831374fa6","familyId":"catalog_family_161771f831374fa6","name":"WebDev Arena","oneLine":"WebDev Arena is a leaderboard for evaluating AI models on web development tasks, including zero-shot generation, complex prompts, and interactive web UI creation. Models are ranked using Elo ratings based on their performance in coding and web development challenges.","description":"WebDev Arena is a leaderboard for evaluating AI models on web development tasks, including zero-shot generation, complex prompts, and interactive web UI creation. Models are ranked using Elo ratings based on their performance in coding and web development challenges.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Frontend Development","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/webdev-arena","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_161771f831374fa6"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/webdev-arena"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"webdev-arena","url":"https://llm-stats.com/benchmarks/webdev-arena","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","frontend development","code"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_webdev-skills-bench_00be1a53","familyId":"bmf_4a4dced0ed77","name":"WebDev-Skills-Bench","oneLine":"A benchmark for evaluating agent skills in web development, comparing matched conditions with length-matched controls and ablation studies on 31 skills across 50 projects.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.23067v1","pdf":"https://arxiv.org/pdf/2608.23067v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We introduce WebDev-Skills-Bench and use it for a controlled empirical study of 31 public WebDev Skills on 50 Web-Bench projects and 1,000 ordered tasks.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.23067"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark for evaluating agent skills in web development, comparing matched conditions with length-matched controls and ablation studies on 31 skills across 50 projects.","whyItMatters":"Raises the standard for agent-skill evaluation by incorporating length-matched controls and per-model audits, revealing that skill efficacy varies by model and deployment.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"e17a7a7b0240e4059a6c34ad5c337543bf21da2dd072dd495f2fb2382f994798"},"motivation":"Agent Skills are reusable procedural modules that are increasingly injected into coding-agent sessions to encode framework conventions, anti-patterns, and reusable tools.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"Formally named benchmark with a repeatable protocol and public tasks, used to compare models and skill conditions.","canonicalNameSource":"abstract","canonicalNameEvidence":"We introduce WebDev-Skills-Bench and use it for a controlled empirical study"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.23067v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":35,"confidence":"Low","horizon":"7d","reason":"Relevant to coding agents and AI-assisted development, but lack of public benchmarks and data limits early attention."},"evaluationMode":"score_submission","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_webigbench_7eee6092","familyId":"bmf_c2c33131ad91","name":"WebIGBench","oneLine":"WebIGBench evaluates code generation for interactive webpages with 103 complex examples and 871 distinct actions, proposing an automated evaluation pipeline for interactive consistency.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":["Code generation"],"topics":["Multimodal","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-29","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2606.00154","pdf":"https://arxiv.org/pdf/2606.00154","project":null,"code":"https://github.com/anoa12159-hue/WebIGBench_eval","data":null,"hfPaper":"https://huggingface.co/papers/2606.00154"},"evidence":{"snippet":"To address these limitations, we introduce WebIGBench, the first benchmark designed to evaluate code generation for interactive webpages with complex interactions.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.00154"},"ranking":{},"description":"WebIGBench evaluates code generation for interactive webpages with 103 complex examples and 871 distinct actions, proposing an automated evaluation pipeline for interactive consistency.","whyItMatters":"Fills the gap in benchmarking interactive webpage code generation, offering a public dataset and evaluation method for model comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"02b84a3e54e2205d58f6e2672fdbbf08df43362a1a96f92846c3f9607760912d"},"motivation":"Recent advancements in multimodal large language models (MLLMs) have achieved remarkable progress in multimodal reasoning and code generation, catalyzing a new paradigm for front-end development.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.00154","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception","Coding & Software Engineering"],"domainScope":"general"},{"id":"bm_webretriever_b067311d","familyId":"bmf_e07e1bfcf561","name":"WebRetriever","oneLine":"Introduces a benchmark with 800 websites and 1,550 tasks for web agent evaluation, plus the NavEval LLM-as-Judge framework and three evaluation protocols.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-07","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.06118","pdf":"https://arxiv.org/pdf/2607.06118","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.06118"},"evidence":{"snippet":"To address these limitations, we introduce WebRetriever, a large-scale benchmark encompassing 800 websites and 1,550 tasks across diverse domains, including consumer, professional, and enterprise sectors, with comprehensive coverage of user intent patterns.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.06118"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Introduces a benchmark with 800 websites and 1,550 tasks for web agent evaluation, plus the NavEval LLM-as-Judge framework and three evaluation protocols.","whyItMatters":"Could offer large-scale cross-domain assessment for web agents, but the evaluation methodology relies on LLM-as-Judge and lacks clear public implementation details.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"eccbb4acb7b58abcaf0259cf698d04a1a0c929277566b685b603967157e82e82"},"motivation":"As web agents increasingly demonstrate capabilities in automated task execution, the development of robust evaluation frameworks for assessing their navigation and task completion performance has emerged as a critical research priority.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.06118","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_7532f2cb612216a0","familyId":"catalog_family_7532f2cb612216a0","name":"WebVoyager","oneLine":"WebVoyager evaluates an agent's ability to navigate and complete tasks on real websites by perceiving page screenshots and executing browser actions.","description":"WebVoyager evaluates an agent's ability to navigate and complete tasks on real websites by perceiving page screenshots and executing browser actions.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Agents","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/vlm/glm-5v-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7532f2cb612216a0"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/webvoyager"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/webvoyager"}],"catalogSources":[{"catalog":"benchlm","sourceId":"webVoyager","url":"https://benchlm.ai/benchmarks/webvoyager","paperUrl":"https://docs.z.ai/guides/vlm/glm-5v-turbo","year":"2026","fullName":"WebVoyager","format":"Interactive browser-agent evaluation","tasks":"Live website workflows","successorKey":null},{"catalog":"llm-stats","sourceId":"webvoyager","url":"https://llm-stats.com/benchmarks/webvoyager","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","agents","vision"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_weclawarena_f743496b","familyId":"bmf_4cf9742a8ee1","name":"WeClawArena","oneLine":"WeClawArena is a benchmark and runtime sandbox for multi-party owned-agent collaboration over personal workspaces. It contains 124 base tasks across six domains, expanded into 620 scenario variants with benign and attack-vector conditions, and reports task success and attack success separately.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.03499","pdf":"https://arxiv.org/pdf/2608.03499","project":null,"code":"https://github.com/kingofspace0wzz/WeClawArena","data":null,"hfPaper":"https://huggingface.co/papers/2608.03499"},"evidence":{"snippet":"We introduce WeClawArena, an auditable benchmark and runtime sandbox for multi-party owned-agent collaboration over personal workspaces.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":8,"hfDailySubmittedAt":"2026-08-11T00:00:00.000Z","githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.03499"},"ranking":{"30d":{"score":25,"rank":102,"coverage":0.85,"confidence":"High"},"90d":{"score":26,"rank":292,"coverage":0.7,"confidence":"Medium"}},"description":"WeClawArena is a benchmark and runtime sandbox for multi-party owned-agent collaboration over personal workspaces. It contains 124 base tasks across six domains, expanded into 620 scenario variants with benign and attack-vector conditions, and reports task success and attack success separately.","whyItMatters":"Existing agent benchmarks do not provide an end-to-end sandbox for verifiable cross-user agent collaboration with realistic digital workspaces. WeClawArena enables evaluation of both collaborative task utility and security risks in human-centered agent networks, supporting diagnosis of privacy leakage, poisoned evidence, and invalid authority paths.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"2a0bad212dc0d7814ba1d3019a723e8879bca65b589da7017ae5b891d1371589"},"motivation":"Recent advances in persistent personal-agent frameworks are making human-centered agent networks realistic deployment targets: each user can be served by an AI agent that acts on the user's behalf, maintains state, and communicates with other agents through social and task relations.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.03499","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"kingofspace0wzz","organizationType":"community","sourceUrl":"https://github.com/kingofspace0wzz/WeClawArena","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_wegenbench_4908a1e1","familyId":"bmf_dddc6369e5dc","name":"WeGenBench","oneLine":"WeGenBench evaluates text-to-image generation across 4,000 bilingual prompts, with multi-dimensional tags and novel VLM-based metrics. It assesses generation quality on scene classification and specific sub-categories.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-18","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.20100","pdf":"https://arxiv.org/pdf/2606.20100","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.20100"},"evidence":{"snippet":"To address these limitations, we propose WeGenBench, a novel benchmark designed for the comprehensive, multi-perspective evaluation of text-to-image generation capabilities.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.20100"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"WeGenBench evaluates text-to-image generation across 4,000 bilingual prompts, with multi-dimensional tags and novel VLM-based metrics. It assesses generation quality on scene classification and specific sub-categories.","whyItMatters":"Existing text-to-image benchmarks lack fine-grained diagnostics. WeGenBench provides a structured resource to pinpoint model deficiencies across dimensions, guiding optimization.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c3a181f87684cd5339acae986fd52e16b1baa35686d2695c45a7d7921ceb8cee"},"motivation":"Recent text-to-image generation models have demonstrated remarkable capabilities in synthesizing highly realistic images from text inputs alone.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.20100","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"WeGenBench Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.20100","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_6c6a0665904b9a6f","familyId":"catalog_family_6c6a0665904b9a6f","name":"WeirdML","oneLine":"A machine-learning engineering benchmark that tests whether LLMs can train models on novel datasets, write PyTorch code, and improve through iterative feedback.","description":"A machine-learning engineering benchmark that tests whether LLMs can train models on novel datasets, write PyTorch code, and improve through iterative feedback.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["External"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://htihle.github.io/weirdml.html","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_6c6a0665904b9a6f"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/weirdml"}],"catalogSources":[{"catalog":"benchlm","sourceId":"weirdMl","url":"https://benchlm.ai/benchmarks/weirdml","paperUrl":"https://htihle.github.io/weirdml.html","year":"2026","fullName":"WeirdML v2","format":"Average accuracy across tasks","tasks":"17 novel ML engineering tasks","successorKey":null}],"catalogCategories":["external"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_wesce_79f8889e","familyId":"bmf_8411e00796ba","name":"WeSCE","oneLine":"WeSCE is a benchmark for quantifying security drift in LLM-driven code editing. It consists of 400 executable programs derived from real-world code, covering feature addition, removal, bug fixing, and refactoring. The benchmark proposes a continuous risk representation and drift measures that capture changes in overall risk, worst-case severity, and vulnerability distribution.","area":"Language & Knowledge","applicationDomains":["Cybersecurity"],"primaryDomain":"Cybersecurity","industrySectors":["Cybersecurity"],"capabilities":[],"topics":["cs.CR"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-15","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.15092","pdf":"https://arxiv.org/pdf/2608.15092","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.15092"},"evidence":{"snippet":"In this work, we introduce WeSCE, a benchmark for quantifying security drift in code editing under weak-security constraints, where tasks specify only functional objectives without explicit security requirements.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15092"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"WeSCE is a benchmark for quantifying security drift in LLM-driven code editing. It consists of 400 executable programs derived from real-world code, covering feature addition, removal, bug fixing, and refactoring. The benchmark proposes a continuous risk representation and drift measures that capture changes in overall risk, worst-case severity, and vulnerability distribution.","whyItMatters":"Code editing with weak-security constraints can introduce vulnerabilities, but existing evaluations lack a systematic measure. WeSCE offers a standardized way to quantify security drift, potentially aiding in selecting models and prompts that minimize security risks during code modifications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"5e413f7937e8cce6dd18cbbee3a2fa72ece0340558e02f94ac0d2ffdbc41c6d1"},"motivation":"In this work, we introduce WeSCE, a benchmark for quantifying security drift in code editing under weak-security constraints, where tasks specify only functional objectives without explicit security requirements.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15092","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_the-illusion-of-textit-what-if-evaluating-_2b6da5c9","familyId":"bmf_f5f7f2b6869b","name":"WhatIfBench","oneLine":"A diagnostic benchmark of 220 open-domain what-if questions across STEM, HSS, and Hybrid scenarios, evaluated with PRISM metrics on causal graphs and explanatory adequacy.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-28","firstSeenAt":"2026-08-31","recognitionConfidence":0.7,"links":{"report":"https://arxiv.org/abs/2608.27953","pdf":"https://arxiv.org/pdf/2608.27953","project":null,"code":"https://github.com/zju-gt/WhatIfBench","data":null,"hfPaper":null},"evidence":{"snippet":"To this end, we present $\\textbf{WhatIfBench}$, a diagnostic benchmark for open-domain, open-form, long-horizon counterfactual causal reasoning, containing 220 what-if questions across STEM, HSS, and Hybrid scenarios.","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":1,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.27953"},"ranking":{"30d":{"score":28,"rank":84,"coverage":0.55,"confidence":"Low"},"90d":{"score":28,"rank":254,"coverage":0.55,"confidence":"Low"}},"description":"A diagnostic benchmark of 220 open-domain what-if questions across STEM, HSS, and Hybrid scenarios, evaluated with PRISM metrics on causal graphs and explanatory adequacy.","whyItMatters":"Exposes the gap between fluent counterfactual narrative and sound causal process reasoning, enabling more rigorous assessment of complex reasoning in LLMs.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-31T04:51:30.440525Z","inputHash":"167d57bc732accc97ce38c49da4d1ef62c640a77b6806deac58dbd704e96f7f9"},"motivation":"Counterfactual reasoning requires models to reason beyond the observed world and explain how altered conditions propagate through downstream consequences.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-31T04:51:30.440525Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is formally defined with a public repository and paper, offering a reusable set of questions and an evaluation framework.","canonicalNameSource":"abstract","canonicalNameEvidence":"we present $\\textbf{WhatIfBench}$, a diagnostic benchmark for open-domain, open-form, long-horizon counterfactual causal reasoning, containing 220 what-if questions across STEM, HSS, and Hybrid scenarios."},"publication":{"status":"acceptance_claimed","venue":"EMNLP 2026","evidence":"Accepted by EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2608.27953","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T01:03:30.163531Z"},"venueAttempts":[{"venueName":"EMNLP 2026","reviewStatus":"accepted","decisionRaw":"Accepted by EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.27953","observedAt":"2026-08-31T01:03:30.163531Z","rawValue":"Accepted by EMNLP 2026","level":"author-claim"}]}],"attentionForecast":{"score":75,"confidence":"Medium","horizon":"7d","reason":"The accepted EMNLP paper introduces a novel counterfactual reasoning benchmark with a new metric, likely attracting broad interest from NLP and reasoning communities."},"evaluationMode":"public_reusable","displayEligible":true,"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_whisperbench_c407b81a","familyId":"bmf_253cb261e509","name":"WhisperBench","oneLine":"WhisperBench evaluates stealth memory injection attacks on persistent personal agents through a 108-case benchmark spanning five risk categories with fact and preference poisoning, using an IMAP/SMTP workflow and an email agent skill.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CR"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-06","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2607.05189","pdf":"https://arxiv.org/pdf/2607.05189","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.05189"},"evidence":{"snippet":"We introduce WhisperBench, a 108-case benchmark spanning five risk categories and both fact and preference poisoning.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.05189"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"WhisperBench evaluates stealth memory injection attacks on persistent personal agents through a 108-case benchmark spanning five risk categories with fact and preference poisoning, using an IMAP/SMTP workflow and an email agent skill.","whyItMatters":"The benchmark addresses the evaluation gap in assessing security of persistent memory in AI agents, providing a way to measure susceptibility to memory injection attacks and the effectiveness of defenses.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"9c6e2c71dfc3405506af7498227236229562bc3cde7c9aa07c021e261ee685f9"},"motivation":"Persistent personal agents combine long-term memory with access to users' external environments, enabling personalized foreground assistance and proactive background execution.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.05189","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_who-when-pro_d21cef2d","familyId":"bmf_164e8ad33d5b","name":"Who&When Pro","oneLine":"Who&When Pro is a benchmark for automated failure attribution in agentic systems, containing 12,326 failed trajectories with golden labels across 3 modalities and 26 benchmarks. It uses a controlled pipeline that injects failures after replaying successful prefixes.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.09996","pdf":"https://arxiv.org/pdf/2607.09996","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.09996"},"evidence":{"snippet":"We introduce Who&When Pro, a large-scale benchmark for automated failure attribution in agentic systems.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.09996"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Who&When Pro is a benchmark for automated failure attribution in agentic systems, containing 12,326 failed trajectories with golden labels across 3 modalities and 26 benchmarks. It uses a controlled pipeline that injects failures after replaying successful prefixes.","whyItMatters":"As agents become more capable, automated failure attribution is crucial for debugging and safety. This benchmark provides a large-scale evaluation to guide the development of systems that can identify where and why failures occur.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"adbcb813bc5a8b17661221c59657a23bb9bfcf7b3d09e35853ba77e206e65ab8"},"motivation":"Automated failure attribution uses LLMs to identify where and why agentic systems fail.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.09996","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_19876f7f9ed94bec","familyId":"catalog_family_19876f7f9ed94bec","name":"WideResearch","oneLine":"A broad research-agent benchmark for open-ended information gathering, synthesis, and answer construction across wide search spaces.","description":"A broad research-agent benchmark for open-ended information gathering, synthesis, and answer construction across wide search spaces.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://qwen.ai/blog?id=qwen3.6","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_19876f7f9ed94bec"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/wideresearch"}],"catalogSources":[{"catalog":"benchlm","sourceId":"wideResearch","url":"https://benchlm.ai/benchmarks/wideresearch","paperUrl":"https://qwen.ai/blog?id=qwen3.6","year":"2026","fullName":"WideResearch","format":"Multi-source research evaluation","tasks":"Open-ended research tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_d91422f7df97f946","familyId":"catalog_family_d91422f7df97f946","name":"WideSearch","oneLine":"WideSearch is an agentic search benchmark that evaluates models' ability to perform broad, parallel search operations across multiple sources. It tests wide-coverage information retrieval and synthesis capabilities.","description":"WideSearch is an agentic search benchmark that evaluates models' ability to perform broad, parallel search operations across multiple sources. It tests wide-coverage information retrieval and synthesis capabilities.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Search","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/widesearch","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d91422f7df97f946"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/widesearch"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"widesearch","url":"https://llm-stats.com/benchmarks/widesearch","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","search","agents"],"catalogModelCount":10,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents","Search & Retrieval"],"domainScope":"general"},{"id":"catalog_60336cacc4ba5437","familyId":"catalog_family_60336cacc4ba5437","name":"WildBench","oneLine":"WildBench is an automated evaluation framework that benchmarks large language models using 1,024 challenging, real-world tasks selected from over one million human-chatbot conversation logs. It introduces two evaluation metrics (WB-Reward and WB-Score) that achieve high correlation with human preferences and uses task-specific checklists for systematic evaluation.","description":"WildBench is an automated evaluation framework that benchmarks large language models using 1,024 challenging, real-world tasks selected from over one million human-chatbot conversation logs. It introduces two evaluation metrics (WB-Reward and WB-Score) that achieve high correlation with human preferences and uses task-specific checklists for systematic evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General","Communication"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2406.04770","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_60336cacc4ba5437"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/wildbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/wild-bench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"wildBench","url":"https://benchlm.ai/benchmarks/wildbench","paperUrl":"https://arxiv.org/abs/2406.04770","year":"2024","fullName":"WildBench","format":"Real-world task evaluation","tasks":"1,024 real-world tasks","successorKey":null},{"catalog":"llm-stats","sourceId":"wild-bench","url":"https://llm-stats.com/benchmarks/wild-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","communication"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_98f32ff7d2707501","familyId":"catalog_family_98f32ff7d2707501","name":"WildClawBench","oneLine":"WildClawBench is an agentic coding benchmark from InternLM/Claw-Eval that reports overall model performance on real-world tool-using development tasks.","description":"WildClawBench is an agentic coding benchmark from InternLM/Claw-Eval that reports overall model performance on real-world tool-using development tasks.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents","Coding"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/wildclawbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_98f32ff7d2707501"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/wildclawbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"wildclawbench","url":"https://llm-stats.com/benchmarks/wildclawbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agents","coding"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_wildhandbench_d9a9447e","familyId":"bmf_632f49955954","name":"WildHandBench","oneLine":"A benchmark containing 500 handwritten documents across three structures, four languages, and nine real-world scenarios, with a Prior-Driven Error metric to quantify reliance on language priors.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-08-24","firstSeenAt":"2026-08-25","recognitionConfidence":1.0,"links":{"report":"http://arxiv.org/abs/2608.22959v1","pdf":"https://arxiv.org/pdf/2608.22959v1","project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"We present WildHandBench, a benchmark containing 500 handwritten documents across three structures (free text, tables, formulas), four languages, and nine real-world scenarios.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-26","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.22959"},"ranking":{"today":{"score":48,"rank":10,"coverage":0.4,"confidence":"Low"},"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A benchmark containing 500 handwritten documents across three structures, four languages, and nine real-world scenarios, with a Prior-Driven Error metric to quantify reliance on language priors.","whyItMatters":"Fills the gap in handwritten document understanding evaluation, revealing systematic prior-driven errors in models compared to humans, which conventional accuracy metrics miss.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-27T04:22:25.189955Z","inputHash":"c7c6345df52436f56a06775262c71d6b130fec1c23030ef1f99932e138646b7c"},"motivation":"While the top model on OmniDocBench now reaches 96.34% overall on printed-document parsing, the ability of current models to handle challenging handwritten documents remains largely uncharacterized.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-name-audit-deferred","reviewedAt":"2026-08-27T04:18:56.318436Z","model":"deepseek-v4-pro","decisionReason":"The benchmark is clearly defined with a stable metric, but the abstract does not state public availability of the dataset or code, weakening immediate reuse potential.","canonicalNameSource":"abstract","canonicalNameEvidence":"We present WildHandBench, a benchmark containing 500 handwritten documents"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"http://arxiv.org/abs/2608.22959v1","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-25T16:21:05.134052Z"},"attentionForecast":{"score":50,"confidence":"Low","horizon":"7d","reason":"Appeals to document AI and multimodal research, but absence of publicly shared artifacts moderates initial attention."},"evaluationMode":"score_submission","displayEligible":false,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_wildtrace_951518d8","familyId":"bmf_31889bba386b","name":"WildTrace","oneLine":"WildTrace evaluates long-context reasoning over naturally occurring long-form sources, focusing on integrating evidence dispersed across distant passages. It includes 481 tasks over 214 sources with seven source-internal evidence geometries, with multi-stage validation ensuring answer groundedness and clue necessity.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Long Context","Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-10","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.09328","pdf":"https://arxiv.org/pdf/2607.09328","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.09328"},"evidence":{"snippet":"We introduce WILDTRACE, a benchmark of 481 tasks over 214 naturally occurring long-form sources such as technical incident reports and lesser-known literary narratives, where all evidence trails arise from the document's own causal, temporal, and narrative logic.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.09328"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"WildTrace evaluates long-context reasoning over naturally occurring long-form sources, focusing on integrating evidence dispersed across distant passages. It includes 481 tasks over 214 sources with seven source-internal evidence geometries, with multi-stage validation ensuring answer groundedness and clue necessity.","whyItMatters":"Existing long-context benchmarks rely on synthetic evidence that may not reflect real source-internal integration. WildTrace provides a more naturalistic evaluation of analytical reading, addressing a gap in assessing true source reasoning.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"83be741e07b82a5f872a01d13afb68a5999c14faa723c202c069d955a2f4b018"},"motivation":"Answering complex questions over long documents frequently requires integrating evidence that the source itself disperses naturally across distant passages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.09328","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning","Long Context & Memory"],"domainScope":"general"},{"id":"catalog_031cd2068b79e93e","familyId":"catalog_family_031cd2068b79e93e","name":"WinoGrande","oneLine":"WinoGrande: An Adversarial Winograd Schema Challenge at Scale. A large-scale dataset of 44,000 pronoun resolution problems designed to test machine commonsense reasoning. Uses adversarial filtering to reduce spurious biases and provides a more robust evaluation of whether AI systems truly understand commonsense or exploit statistical shortcuts. Current best AI methods achieve 59.4-79.1% accuracy, significantly below human performance of 94.0%.","description":"WinoGrande: An Adversarial Winograd Schema Challenge at Scale. A large-scale dataset of 44,000 pronoun resolution problems designed to test machine commonsense reasoning. Uses adversarial filtering to reduce spurious biases and provides a more robust evaluation of whether AI systems truly understand commonsense or exploit statistical shortcuts. Current best AI methods achieve 59.4-79.1% accuracy, significantly below human performance of 94.0%.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","Language"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_031cd2068b79e93e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/winogrande"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/winogrande"}],"catalogSources":[{"catalog":"benchlm","sourceId":"winogrande","url":"https://benchlm.ai/benchmarks/winogrande","paperUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf","year":"2026","fullName":"WinoGrande","format":"Exact match","tasks":"Coreference resolution questions","successorKey":null},{"catalog":"llm-stats","sourceId":"winogrande","url":"https://llm-stats.com/benchmarks/winogrande","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","language"],"catalogModelCount":22,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_wirecraft_e44247e4","familyId":"bmf_9d7bed8a19bb","name":"WireCraft","oneLine":"WireCraft is a simulation benchmark for industrial deformable linear object manipulation, featuring three task families (connector insertion, clip routing, channel seating) with configurable difficulty, two DLO physics models, and shared evaluation metrics for RL, IL, and VLA policies across simulation and physical UR5 trajectories.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":["Robot manipulation"],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-16","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18097","pdf":"https://arxiv.org/pdf/2606.18097","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18097"},"evidence":{"snippet":"To bridge this gap, we introduce WireCraft, a simulation benchmark for industrial DLO manipulation with configurable difficulty and assets, spanning three task families: connector insertion, clip routing, and channel seating.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18097"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"WireCraft is a simulation benchmark for industrial deformable linear object manipulation, featuring three task families (connector insertion, clip routing, channel seating) with configurable difficulty, two DLO physics models, and shared evaluation metrics for RL, IL, and VLA policies across simulation and physical UR5 trajectories.","whyItMatters":"WireCraft addresses the lack of benchmarks that combine industrial fixtures, configurable tasks, and shared evaluation protocols for deformable objects, enabling reproducible comparison and progress in vision-based policy learning for industrial assembly.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"62e32d6efa59a285a09a58aa3372fddf3a04567a481033ba88cfb59363732e65"},"motivation":"Deformable Linear Objects (DLOs), such as wires and cables, are central to industrial assembly.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.18097","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"WireCraft Contributors","organizationType":"benchmark-organization","sourceUrl":"https://arxiv.org/abs/2606.18097","role":"benchmark-publisher"}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"catalog_31909bc8f19a3e93","familyId":"catalog_family_31909bc8f19a3e93","name":"WMDP","oneLine":"Weapons of Mass Destruction (WMDP) is a multiple-choice benchmark on dual-use biology, chemistry, and cyber knowledge. It measures a model's capacity to enable malicious actors to design, synthesize, acquire, or use chemical, biological, radiological, or nuclear (CBRN) weapons.","description":"Weapons of Mass Destruction (WMDP) is a multiple-choice benchmark on dual-use biology, chemistry, and cyber knowledge. It measures a model's capacity to enable malicious actors to design, synthesize, acquire, or use chemical, biological, radiological, or nuclear (CBRN) weapons.","area":"Safety & Trustworthiness","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Safety","Healthcare","Biology","Chemistry"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/wmdp","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_31909bc8f19a3e93"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/wmdp"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"wmdp","url":"https://llm-stats.com/benchmarks/wmdp","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety","healthcare","biology","chemistry"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"specific"},{"id":"catalog_d5fdceb2154bab3d","familyId":"catalog_family_d5fdceb2154bab3d","name":"WMT23","oneLine":"The Eighth Conference on Machine Translation (WMT23) benchmark evaluating machine translation systems across 8 language pairs (14 translation directions) including general, biomedical, literary, and low-resource language translation tasks. Features specialized shared tasks for quality estimation, metrics evaluation, sign language translation, and discourse-level literary translation with professional human assessment.","description":"The Eighth Conference on Machine Translation (WMT23) benchmark evaluating machine translation systems across 8 language pairs (14 translation directions) including general, biomedical, literary, and low-resource language translation tasks. Features specialized shared tasks for quality estimation, metrics evaluation, sign language translation, and discourse-level literary translation with professional human assessment.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences"],"primaryDomain":"Health & Life Sciences","industrySectors":[],"capabilities":[],"topics":["Language","Healthcare"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/wmt23","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_d5fdceb2154bab3d"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/wmt23"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"wmt23","url":"https://llm-stats.com/benchmarks/wmt23","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","healthcare"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"catalog_7ce3657bfc8882e2","familyId":"catalog_family_7ce3657bfc8882e2","name":"WMT24++","oneLine":"WMT24++ is a comprehensive multilingual machine translation benchmark that expands the WMT24 dataset to cover 55 languages and dialects. It includes human-written references and post-edits across four domains (literary, news, social, and speech) to evaluate machine translation systems and large language models across diverse linguistic contexts.","description":"WMT24++ is a comprehensive multilingual machine translation benchmark that expands the WMT24 dataset to cover 55 languages and dialects. It includes human-written references and post-edits across four domains (literary, news, social, and speech) to evaluate machine translation systems and large language models across diverse linguistic contexts.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/wmt24++","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7ce3657bfc8882e2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/wmt24++"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"wmt24++","url":"https://llm-stats.com/benchmarks/wmt24++","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language"],"catalogModelCount":23,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_workflow-gym_72fcdcd4","familyId":"bmf_bc38808cdab5","name":"Workflow-GYM","oneLine":"Workflow-GYM evaluates AI agents on long-horizon GUI tasks in professional domains, using specialized software environments and economically valuable workflows. Tasks require end-to-end operation of graphical user interfaces, with success rates measured by task completion.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11042","pdf":"https://arxiv.org/pdf/2606.11042","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.11042"},"evidence":{"snippet":"To bridge this gap, we introduce Workflow-GYM, a benchmark for long-horizon GUI tasks centered on professional domains and specialized software environments.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":22,"hfDailySubmittedAt":"2026-06-10T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.11042"},"ranking":{"90d":{"score":51,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"Workflow-GYM evaluates AI agents on long-horizon GUI tasks in professional domains, using specialized software environments and economically valuable workflows. Tasks require end-to-end operation of graphical user interfaces, with success rates measured by task completion.","whyItMatters":"Most GUI benchmarks cover simple, short-horizon tasks in general software, leaving a gap for professional, long-horizon workflows. Workflow-GYM provides a fixed protocol for measuring agent performance on such tasks, which is valuable as organizations consider deploying agents for complex professional work.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"badacf3a03dd44823e3abf9dd0292873ffea1d6474fd00c6cb1085e6cffd3063"},"motivation":"Recent years have witnessed the rapid evolution of AI agents toward handling increasingly complex, real-world tasks.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11042","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Workflow-GYM Team","organizationType":"academic-lab","sourceUrl":"https://arxiv.org/abs/2606.11042","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_7a2bde6e6f86bc45","familyId":"catalog_family_7a2bde6e6f86bc45","name":"Workspace Bench","oneLine":"Workspace Bench evaluates AI agents on high-economic-value workplace tasks that span multi-step planning, file processing, and tool use across realistic office and productivity workflows.","description":"Workspace Bench evaluates AI agents on high-economic-value workplace tasks that span multi-step planning, file processing, and tool use across realistic office and productivity workflows.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/workspace-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_7a2bde6e6f86bc45"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/workspace-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"workspace-bench","url":"https://llm-stats.com/benchmarks/workspace-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","agents"],"catalogModelCount":3,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_worksurface-bench_7866ac7c","familyId":"bmf_639fa8c3c183","name":"WorkSurface-Bench","oneLine":"WorkSurface-Bench evaluates enterprise agents on knowledge routing across documents, tables, and graphs. It includes 1,151 atomic tasks with auditable reference answers and scoring for route, evidence, answer, and efficiency.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Factuality"],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.25765","pdf":"https://arxiv.org/pdf/2607.25765","project":null,"code":"https://github.com/haolpku/WorkSurface-Bench","data":null,"hfPaper":"https://huggingface.co/papers/2607.25765"},"evidence":{"snippet":"We introduce WorkSurface-Bench, a benchmark for evaluating this capability as surface routing.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":7,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.25765"},"ranking":{"90d":{"score":31,"rank":237,"coverage":0.7,"confidence":"Medium"}},"description":"WorkSurface-Bench evaluates enterprise agents on knowledge routing across documents, tables, and graphs. It includes 1,151 atomic tasks with auditable reference answers and scoring for route, evidence, answer, and efficiency.","whyItMatters":"It isolates surface routing from evidence acquisition and answer generation, showing that correct routing is necessary but insufficient. This helps diagnose why agents fail on multi-surface tasks and informs design of routing-aware systems.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"1472e13f84f8ad0f4c160895993807ed362d13fb6b151507225ad3945576d897"},"motivation":"Enterprise agents often need to integrate heterogeneous knowledge sources: documents for narrative facts, tables for computation, and dependency graphs for file relationships.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.25765","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"haolpku","organizationType":"academic-lab","sourceUrl":"https://github.com/haolpku/WorkSurface-Bench","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_worldbench_a7ecd2c0","familyId":"bmf_365c8ee11b3d","name":"WorldBench","oneLine":"WorldBench evaluates multimodal large language models on visually diverse reasoning questions, with accuracy as the metric on a curated dataset.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning"],"topics":["Multimodal","Reasoning"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-04","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.06538","pdf":"https://arxiv.org/pdf/2606.06538","project":"https://worldbench-vl.github.io/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.06538"},"evidence":{"snippet":"We present WorldBench, a challenging and visually diverse reasoning benchmark to evaluate Multimodal Large Language Models (MLLMs).","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":3,"hfDailySubmittedAt":"2026-06-08T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.06538"},"ranking":{"90d":{"score":46,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"WorldBench evaluates multimodal large language models on visually diverse reasoning questions, with accuracy as the metric on a curated dataset.","whyItMatters":"Highlights visual diversity gaps in existing benchmarks; provides a challenging fixed dataset for model comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ad7c7d3a2f428b2672931cdfa4ebc4cfd6b312c59e445e9628866908c7b7007b"},"motivation":"In real-world applications, models are expected to perform reliably across diverse settings.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.06538","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general","catalogSources":[{"catalog":"llm-stats","sourceId":"worldbench","url":"https://llm-stats.com/benchmarks/worldbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["knowledge","multimodal","vision"],"catalogModelCount":2,"catalogStarCount":0},{"id":"bm_worldcoder-bench_8f868de5","familyId":"bmf_46b10b77b482","name":"WorldCoder-Bench","oneLine":"Evaluates autonomous physically grounded 3D world synthesis from natural language, with 2,026 expert-curated tasks across Simulation, Rendering, and Application scenarios, using execution-based verification via StateProbe to check hidden behavioral contracts over runtime states.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.AI"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Inspectable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.01869","pdf":"https://arxiv.org/pdf/2606.01869","project":"https://anonymous.4open.science/r/WorldCoder-Bench/","code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.01869"},"evidence":{"snippet":"We introduce WorldCoder-Bench, a benchmark for autonomous, physically grounded 3D world synthesis.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":2,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.01869"},"ranking":{},"description":"Evaluates autonomous physically grounded 3D world synthesis from natural language, with 2,026 expert-curated tasks across Simulation, Rendering, and Application scenarios, using execution-based verification via StateProbe to check hidden behavioral contracts over runtime states.","whyItMatters":"Existing web-generation benchmarks observe only pixels or DOM nodes, missing the mechanics of Three.js worlds inside canvas elements. This benchmark provides a protocol for verifying hidden contracts, enabling assessment of correctness-adjusted cost and time efficiency for 3D world synthesis.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3227ed109d8f98090ac73e1e2cd973521870dd9ffe97c2ac2216641c779ad4bb"},"motivation":"Large language models (LLMs) are increasingly asked not only to write static interfaces, but to construct executable interactive worlds from natural language.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.01869","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"WorldCoder-Bench team","organizationType":"academic-lab","sourceUrl":"https://anonymous.4open.science/r/WorldCoder-Bench/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_worldcuparena_3034c352","familyId":"bmf_4b5c0cb0434c","name":"WorldCupArena","oneLine":"WorldCupArena evaluates language models and deep-research agents on football forecasting across multiple tasks including result, score, player events, statistics, and tournament outcomes. It uses a composite score and supports adding new schedules for future events.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.18084","pdf":"https://arxiv.org/pdf/2607.18084","project":null,"code":"https://github.com/wzk1015/WorldCupArena","data":null,"hfPaper":"https://huggingface.co/papers/2607.18084"},"evidence":{"snippet":"We present WorldCupArena, a dynamic benchmark for language models and deep-research agents.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":null,"githubStars":24,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.18084"},"ranking":{"90d":{"score":44,"rank":118,"coverage":0.7,"confidence":"Medium"}},"description":"WorldCupArena evaluates language models and deep-research agents on football forecasting across multiple tasks including result, score, player events, statistics, and tournament outcomes. It uses a composite score and supports adding new schedules for future events.","whyItMatters":"This benchmark provides a dynamic, real-world testbed for evaluating models on multi-source reasoning and structured prediction with ground truth on a fixed schedule. It offers practical value in comparing model performance against market and human baselines, revealing differences in detailed predictions beyond result accuracy.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"3abb2ecf948b0e807d56a9402f9390a5f6a48869a5ecdabf830ae8e4c4f294ec"},"motivation":"Predicting a football match before kickoff requires more than knowing past results: a model must use changing information and make a clear prediction before the answer is available.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.18084","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_worldexam_7f2f166e","familyId":"bmf_3931c2ca2952","name":"WorldExam","oneLine":"WorldExam evaluates controllable video generation models as world models across four levels: Visual Quality, Control Adherence, Spatial Consistency, and World Reactivity, with 1,474 cases across eight tasks.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.02603","pdf":"https://arxiv.org/pdf/2608.02603","project":"https://WorldExam.github.io","code":"https://github.com/YuxueYang1204/worldexam","data":null,"hfPaper":"https://huggingface.co/papers/2608.02603"},"evidence":{"snippet":"We introduce WorldExam, a hierarchical diagnostic benchmark spanning four levels: Visual Quality, Control Adherence, Spatial Consistency, and World Reactivity.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":34,"hfDailySubmittedAt":"2026-08-04T00:00:00.000Z","githubStars":22,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.02603"},"ranking":{"30d":{"score":52,"rank":22,"coverage":0.85,"confidence":"High"},"90d":{"score":48,"rank":86,"coverage":0.7,"confidence":"Medium"}},"description":"WorldExam evaluates controllable video generation models as world models across four levels: Visual Quality, Control Adherence, Spatial Consistency, and World Reactivity, with 1,474 cases across eight tasks.","whyItMatters":"Fills the gap in evaluating inherent world reactivity beyond visual quality and instruction fulfillment, providing a hierarchical diagnostic for model comparison.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"ad6f7e1c7fa7d00cdeb16255b1471a37c16e556e55fa1f1cef5b11b412f7da0e"},"motivation":"Controllable video generation models are increasingly being developed as world models.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.02603","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_worldlines_ada0e776","familyId":"bmf_38b4fbd5fcdb","name":"WorldLines","oneLine":"WorldLines evaluates long-horizon stateful embodied agents in household environments through Memory QA and Embodied Task Planning, using temporally extended household traces with dialogues and state changes.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.18847","pdf":"https://arxiv.org/pdf/2606.18847","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.18847"},"evidence":{"snippet":"We introduce WorldLines, a project-driven benchmark for long-horizon embodied household assistance.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":5,"hfDailySubmittedAt":"2026-06-22T00:00:00.000Z","githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.18847"},"ranking":{"90d":{"score":47,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"WorldLines evaluates long-horizon stateful embodied agents in household environments through Memory QA and Embodied Task Planning, using temporally extended household traces with dialogues and state changes.","whyItMatters":"Existing benchmarks lack evaluation of long-term memory in dynamic embodied settings; WorldLines fills this gap by testing both memory retrieval and planning over extended interactions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"77c9ca12db8cf032d8beb481f9491e7bb0eedf978944ef78b14ad27ae2882300"},"motivation":"To assist humans over extended periods in real homes, embodied agents must remember user routines, world states, and past interactions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"EMNLP 2026","evidence":"Accepted to EMNLP 2026","evidenceUrl":"https://arxiv.org/abs/2606.18847","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"EMNLP 2026","reviewStatus":"accepted","decisionRaw":"Accepted to EMNLP 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2606.18847","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to EMNLP 2026","level":"author-claim"}]}],"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_worldolympiad_08793185","familyId":"bmf_9371008df044","name":"WorldOlympiad","oneLine":"WorldOlympiad is a triathlon-style benchmark for video-based world models, evaluating physical faithfulness, geometric consistency, and interaction fidelity across gaming, robotics, and general real-world scenarios. It includes 1,000 long videos and interpretable automatic metrics.","area":"Vision & 3D","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-09","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.11129","pdf":"https://arxiv.org/pdf/2606.11129","project":"https://alibaba-damo-academy.github.io/WorldOlympiad/","code":"https://github.com/alibaba-damo-academy/WorldOlympiad","data":null,"hfPaper":"https://huggingface.co/papers/2606.11129"},"evidence":{"snippet":"We introduce WorldOlympiad, a benchmark for diagnosing video-based world models across physical faithfulness, geometric consistency, and interaction fidelity.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":32,"hfDailySubmittedAt":"2026-06-10T00:00:00.000Z","githubStars":56,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.11129"},"ranking":{"90d":{"score":55,"rank":37,"coverage":0.7,"confidence":"Medium"}},"description":"WorldOlympiad is a triathlon-style benchmark for video-based world models, evaluating physical faithfulness, geometric consistency, and interaction fidelity across gaming, robotics, and general real-world scenarios. It includes 1,000 long videos and interpretable automatic metrics.","whyItMatters":"Existing video generation benchmarks focus on visual quality or short-term coherence, lacking diagnostics for physical rule compliance, 3D structural consistency, and long-horizon interactive control. WorldOlympiad provides a structured protocol to identify specific failure modes in world models, informing model development and selection for embodied and interactive applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0ba2afb42eaefe4f642c77b35a3a0b41369eaddc1ffc113fb0e3c4f008b915c6"},"motivation":"We introduce WorldOlympiad, a benchmark for diagnosing video-based world models across physical faithfulness, geometric consistency, and interaction fidelity.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.11129","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"Alibaba DAMO Academy","organizationType":"company-research-lab","sourceUrl":"https://alibaba-damo-academy.github.io/WorldOlympiad/","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"bm_worldroambench_ba7d4093","familyId":"bmf_0d99b2d6873b","name":"WorldRoamBench","oneLine":"WorldRoamBench evaluates interactive world models on long-horizon stability across action, vision, physics, and memory dimensions. It includes 600+ test cases in various scenes and views with continuous interaction.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Geometric reasoning"],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-30","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.31672","pdf":"https://arxiv.org/pdf/2606.31672","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.31672"},"evidence":{"snippet":"We introduce WorldRoamBench, an open-world benchmark for long-horizon stability across four dimensions, each with tailored innovations: (i) Action: per-frame action metric bypassing cross-model semantic scale disparity and exposing failures hidden by trajectory; (ii) Vision: segment-based drift metric capturing non-monotonic mid-sequence collapse missed by start-vs-end comparisons; (iii) Physics: controllability-gated evaluation over mechanics, optics, and 3D consistency, scoring plausibility un","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.31672"},"ranking":{"90d":{"score":43,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"WorldRoamBench evaluates interactive world models on long-horizon stability across action, vision, physics, and memory dimensions. It includes 600+ test cases in various scenes and views with continuous interaction.","whyItMatters":"Interactive world models need stable, physically grounded, and memory-faithful behavior, but existing benchmarks ignore these aspects. WorldRoamBench provides a comprehensive evaluation revealing that no current model reliably satisfies all dimensions.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"0151c8c8bd0d6bf6a0a8a303bcdef18953be429dbe8bd12944a57e965bb33af5"},"motivation":"Despite rapid progress in interactive world models (IWMs), existing benchmarks evaluate action following only at trajectory level and ignore memory and interaction physics.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.31672","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_4db4aada87823846","familyId":"catalog_family_4db4aada87823846","name":"WorldVQA","oneLine":"WorldVQA is a benchmark designed to evaluate atomic vision-centric world knowledge. It assesses models' ability to understand and reason about visual elements representing real-world knowledge.","description":"WorldVQA is a benchmark designed to evaluate atomic vision-centric world knowledge. It assesses models' ability to understand and reason about visual elements representing real-world knowledge.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/worldvqa","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_4db4aada87823846"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/worldvqa"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"worldvqa","url":"https://llm-stats.com/benchmarks/worldvqa","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_286c73623c7b2180","familyId":"catalog_family_286c73623c7b2180","name":"WorldVQA ForceAnswer","oneLine":"A forced-answer WorldVQA variant for atomic visual world knowledge.","description":"A forced-answer WorldVQA variant for atomic visual world knowledge.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_286c73623c7b2180"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/worldvqaforceanswer"}],"catalogSources":[{"catalog":"benchlm","sourceId":"worldVqaForceAnswer","url":"https://benchlm.ai/benchmarks/worldvqaforceanswer","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"WorldVQA ForceAnswer","format":"Forced-answer visual QA","tasks":"Atomic visual world-knowledge questions","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_3712248b2b66bf45","familyId":"catalog_family_3712248b2b66bf45","name":"WritingBench","oneLine":"A comprehensive benchmark for evaluating large language models' generative writing capabilities across 6 core writing domains (Academic & Engineering, Finance & Business, Politics & Law, Literature & Art, Education, Advertising & Marketing) and 100 subdomains. Contains 1,239 queries with a query-dependent evaluation framework that dynamically generates 5 instance-specific assessment criteria for each writing task, using a fine-tuned critic model to score responses on style, format, and length dimensions.","description":"A comprehensive benchmark for evaluating large language models' generative writing capabilities across 6 core writing domains (Academic & Engineering, Finance & Business, Politics & Law, Literature & Art, Education, Advertising & Marketing) and 100 subdomains. Contains 1,239 queries with a query-dependent evaluation framework that dynamically generates 5 instance-specific assessment criteria for each writing task, using a fine-tuned critic model to score responses on style, format, and length dimensions.","area":"Language & Knowledge","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Legal","Finance","Communication","Creativity","Writing"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/writingbench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_3712248b2b66bf45"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/writingbench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"writingbench","url":"https://llm-stats.com/benchmarks/writingbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["legal","finance","communication","creativity","writing"],"catalogModelCount":15,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"specific"},{"id":"bm_wsadbench_d5d448b3","familyId":"bmf_44fa01c8bb2a","name":"WSADBench","oneLine":"WSADBench is a benchmark for weakly supervised anomaly detection (WSAD) that unifies evaluation across incomplete, inexact, and inaccurate supervision scenarios. It evaluates 36 algorithms across 4 modalities (tabular, video, image features, text embeddings) by systematically varying label quantity, granularity, and quality, with protocols and code provided.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.LG"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-05-25","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2605.26068","pdf":"https://arxiv.org/pdf/2605.26068","project":null,"code":"https://github.com/SUFE-AILAB/WSADBench","data":null,"hfPaper":"https://huggingface.co/papers/2605.26068"},"evidence":{"snippet":"We release WSADBench as an open-source benchmark with code and datasets to facilitate future WSAD research: https://github.com/SUFE-AILAB/WSADBench.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":8,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.26068"},"ranking":{},"description":"WSADBench is a benchmark for weakly supervised anomaly detection (WSAD) that unifies evaluation across incomplete, inexact, and inaccurate supervision scenarios. It evaluates 36 algorithms across 4 modalities (tabular, video, image features, text embeddings) by systematically varying label quantity, granularity, and quality, with protocols and code provided.","whyItMatters":"The field of weakly supervised anomaly detection has lacked a unified evaluation framework, with existing benchmarks isolating the three supervision types. WSADBench provides a standardized comparison that reveals performance boundaries across scenarios and informs algorithm selection for practitioners facing limited or noisy labels.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"694a3aefd56c44b4310cd6fb08444abd8c2be2bab95e5d467a32ca978aae6048"},"motivation":"Weakly supervised anomaly detection (WSAD) has developed in three primary directions: incomplete, inexact, and inaccurate supervision.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"KDD 2026 Datasets and Benchmarks Track","evidence":"Accepted at KDD 2026 Datasets and Benchmarks Track","evidenceUrl":"https://arxiv.org/abs/2605.26068","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"KDD 2026 Datasets and Benchmarks Track","reviewStatus":"accepted","decisionRaw":"Accepted at KDD 2026 Datasets and Benchmarks Track","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2605.26068","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted at KDD 2026 Datasets and Benchmarks Track","level":"author-claim"}]}],"publishers":[{"name":"SUFE-AILAB","organizationType":"academic-lab","sourceUrl":"https://github.com/SUFE-AILAB/WSADBench","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_wse-bench_2966278a","familyId":"bmf_79b3bfde8025","name":"WSE-bench","oneLine":"WSE-bench evaluates LLM storytelling in open-ended world simulations, assessing three capacities: Generation Coverage (proportion of planned narrative steps), Consistency (canon coherence), and Richness (meaningful development). Uses a process benchmark to compare agent architectures.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-16","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2608.15654","pdf":"https://arxiv.org/pdf/2608.15654","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.15654"},"evidence":{"snippet":"We introduce WSE-bench, a process benchmark that separately evaluates sustained generation, canonical coherence, and meaningful development in dynamic LLM storytelling.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.15654"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"WSE-bench evaluates LLM storytelling in open-ended world simulations, assessing three capacities: Generation Coverage (proportion of planned narrative steps), Consistency (canon coherence), and Richness (meaningful development). Uses a process benchmark to compare agent architectures.","whyItMatters":"Storytelling evaluation has focused on finished stories, but open-ended narratives require sustained generation, coherence, and development. WSE-bench makes these dynamics visible and shows they are distinct capacities.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"518516079b768bba91f6c949337ea956a14b141d39a47727c65f70a5793db472"},"motivation":"Large language models can write fluent stories, but open-ended storytelling requires more than local fluency.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.15654","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_wuicc-bench_284cff3d","familyId":"bmf_edbc54d76868","name":"WUICC-bench","oneLine":"WUICC-bench evaluates image change captioning for web UI visual regression testing, using natural language descriptions of UI changes. Scoring includes caption quality metrics and assesses suppression of non-meaningful noise.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-02","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.01728","pdf":"https://arxiv.org/pdf/2607.01728","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.01728"},"evidence":{"snippet":"To address the gap, we propose a new task, Web UI Image Change Captioning (WUICC), which sits at the intersection of VRT and image difference captioning (IDC), and release WUICC-bench, its first dataset and benchmark for the task.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.01728"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"WUICC-bench evaluates image change captioning for web UI visual regression testing, using natural language descriptions of UI changes. Scoring includes caption quality metrics and assesses suppression of non-meaningful noise.","whyItMatters":"Pixel-level VRT is semantically blind and produces false positives. This benchmark enables development of change captioning systems that describe UI changes in words, improving regression testing efficiency.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-21T04:30:40.934319Z","inputHash":"73003b7c5003a2d8108102f09dfc900e16d0f36f07ed622e92a1e5532d2e96ec"},"motivation":"Visual regression testing (VRT) is a standard quality assurance step in modern software release pipelines.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.01728","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_wuyu-envle-bench_a002dca0","familyId":"bmf_cb77b53e6c10","name":"WuYu-EnvLE-Bench","oneLine":"WuYu-EnvLE-Bench evaluates LLMs on environmental law enforcement with 2,521 instances across 14 tasks and 12 pollution-medium subdomains, covering pre-, in-, and post-enforcement workflows. It uses Absolute Environmental Enforcement Score (AES) and Intelligent Enforcement Index (IEI) for evaluation.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-20","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.17745","pdf":"https://arxiv.org/pdf/2607.17745","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.17745"},"evidence":{"snippet":"We introduce WuYu-EnvLE-Bench, a benchmark built from real enforcement cases, regulatory standards, and expert review.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.17745"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"WuYu-EnvLE-Bench evaluates LLMs on environmental law enforcement with 2,521 instances across 14 tasks and 12 pollution-medium subdomains, covering pre-, in-, and post-enforcement workflows. It uses Absolute Environmental Enforcement Score (AES) and Intelligent Enforcement Index (IEI) for evaluation.","whyItMatters":"The benchmark fills a gap in evaluating LLMs for evidence-grounded, rule-aware reasoning in environmental enforcement. It provides practical assessment of model reliability in traceable decision-making, highlighting limitations in evidence-chain construction and procedural judgment, which is critical for legal and regulatory applications.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f20497a9c0e3d2562c50f6ec3232b598e3967c5a446c8b6d4f08be77c4b3c331"},"motivation":"Large language models (LLMs) are increasingly considered for environmental enforcement, but their ability to produce traceable enforcement decisions remains unclear.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.17745","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_wuyueval_f4ea028f","familyId":"bmf_e2fd7bd79b60","name":"WuYuEval","oneLine":"A multi-level benchmark evaluates LLMs in solid waste management across foundational knowledge, domain reasoning, and expert decision-making using closed-ended and open-ended questions.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":["Reasoning","Factuality"],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-24","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.07529","pdf":"https://arxiv.org/pdf/2608.07529","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.07529"},"evidence":{"snippet":"We introduce WuYuEval, a multi-level benchmark for evaluating LLMs in SWM across foundational knowledge, domain reasoning, and expert decision-making.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.07529"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"A multi-level benchmark evaluates LLMs in solid waste management across foundational knowledge, domain reasoning, and expert decision-making using closed-ended and open-ended questions.","whyItMatters":"Fills a gap in evaluating professional decisions under engineering, environmental, and policy constraints, providing a resource for developing domain-oriented models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c9abf8074a5ea6e32af3d836249e207b330a7244c21e5a6743930774785ce4ae"},"motivation":"Large language models (LLMs) are increasingly used as technical assistants, but their competence in solid waste management (SWM) remains difficult to assess because existing benchmarks emphasize general knowledge rather than professional decisions under engineering, environmental, and policy constraints.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.07529","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_x-slides_283eef58","familyId":"bmf_6e62a789c14b","name":"X+Slides","oneLine":"X+Slides evaluates audience-conditioned slide generation from source documents, using a dynamic framework of 8,133 source-grounded probes across 113 topics and seven presentation scenes, reporting metrics like Audience Coverage and Efficiency.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-06-17","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.19256","pdf":"https://arxiv.org/pdf/2606.19256","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.19256"},"evidence":{"snippet":"To bridge this gap, we introduce X+Slides, a benchmark specifically designed for audience-conditioned slide generation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.19256"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"X+Slides evaluates audience-conditioned slide generation from source documents, using a dynamic framework of 8,133 source-grounded probes across 113 topics and seven presentation scenes, reporting metrics like Audience Coverage and Efficiency.","whyItMatters":"The benchmark addresses the gap in evaluating slide generation by considering target audience, offering metrics that measure coverage of audience-essential information and source grounding, which matters for selecting systems that meet specific audience needs.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"907a485bfc290ad3bddfbc28c89a3d5791acf46cbd4d941410ea0e77a9c0133a"},"motivation":"Automatically generating slide decks from source documents is an important application of large language models (LLMs).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.19256","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_x-stream_b79ec653","familyId":"bmf_229d06cf6ea7","name":"X-Stream","oneLine":"Evaluates multimodal large language models on multi-stream streaming understanding, with 4,220 QA pairs across 932 videos covering 11 subtasks in multi-window, multi-view, and multi-device scenarios, using a dual-verification construction pipeline and online inference under a fixed average video-token rate.","area":"Vision & 3D","applicationDomains":["Transport & Logistics"],"primaryDomain":"Transport & Logistics","industrySectors":["Automotive"],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-06-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2606.02482","pdf":"https://arxiv.org/pdf/2606.02482","project":"https://peiwensun2000.github.io/xstream/","code":"https://github.com/PeiwenSun2000/X-Stream","data":null,"hfPaper":"https://huggingface.co/papers/2606.02482"},"evidence":{"snippet":"To bridge this, we introduce X-Stream, the first benchmark dedicated to multi-stream streaming understanding.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":36,"hfDailySubmittedAt":null,"githubStars":34,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.02482"},"ranking":{},"description":"Evaluates multimodal large language models on multi-stream streaming understanding, with 4,220 QA pairs across 932 videos covering 11 subtasks in multi-window, multi-view, and multi-device scenarios, using a dual-verification construction pipeline and online inference under a fixed average video-token rate.","whyItMatters":"Existing benchmarks focus on single-stream video understanding, leaving a gap for evaluating concurrent video streams common in live sports, autonomous driving, and multi-screen applications. This benchmark provides a practical evaluation protocol for multi-stream reasoning and exposes limitations in current models.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"b1ff0c2b2e664aee9dd1a89b560d911f0249840395675cd01ee014873fe5bb87"},"motivation":"While video streaming understanding has made significant strides, real-world applications, such as live sports broadcasting, autonomous driving, and multi-screen collaboration, inherently demand continuous, multi-stream interactions.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.02482","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"MMLab, CUHK","organizationType":"academic-lab","sourceUrl":"https://github.com/PeiwenSun2000/X-Stream","role":"benchmark-publisher"},{"name":"Huawei Inc.","organizationType":"company-research-lab","sourceUrl":"https://github.com/PeiwenSun2000/X-Stream","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"specific"},{"id":"catalog_645de4b84ba4ee6e","familyId":"catalog_family_645de4b84ba4ee6e","name":"xDailyBench","oneLine":"xDailyBench evaluates AI agents on white-collar office work, covering everyday professional tasks such as document handling, consultation, and multi-step productivity workflows.","description":"xDailyBench evaluates AI agents on white-collar office work, covering everyday professional tasks such as document handling, consultation, and multi-step productivity workflows.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning","General","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/xdailybench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_645de4b84ba4ee6e"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/xdailybench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"xdailybench","url":"https://llm-stats.com/benchmarks/xdailybench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning","general","agents"],"catalogModelCount":2,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"bm_xeworld_c6f125ea","familyId":"bmf_0e0ce96d0a58","name":"XEWorld","oneLine":"XEWorld is a testbed for evaluating cross-embodiment generalization of action-conditioned world models, but it is primarily a research probe without a defined public benchmark protocol.","area":"Robotics & Embodied AI","applicationDomains":["Robotics & Autonomous Systems"],"primaryDomain":"Robotics & Autonomous Systems","industrySectors":["Robotics"],"capabilities":[],"topics":["Robotics"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-08-06","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.05799","pdf":"https://arxiv.org/pdf/2608.05799","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.05799"},"evidence":{"snippet":"To answer whether a model can faithfully render a robot it has never seen, we introduce XEWorld, a controlled cross-embodiment testbed for world models that isolates embodiments by evaluating held-out robots within physically identical scenes.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.05799"},"ranking":{"30d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"},"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"XEWorld is a testbed for evaluating cross-embodiment generalization of action-conditioned world models, but it is primarily a research probe without a defined public benchmark protocol.","whyItMatters":"The study highlights limitations in current world models, but the evaluation setup is tied to the paper's analysis and lacks a standalone comparison path.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"dd575afd8c5009f71301c6debc14c44f0e2cc26970a8b8e9478ab69b2b9bec67"},"motivation":"Action-conditioned world models are promising learned simulators for robotic manipulation, yet evaluating them exclusively on training robots fails to reveal whether they capture physical dynamics or merely memorize visual patterns.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.05799","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Robotics & Embodied Intelligence"],"domainScope":"specific"},{"id":"bm_xih-bench_b0306814","familyId":"bmf_a796a6dc7c9f","name":"XIH-Bench","oneLine":"Evaluates instruction hierarchy compliance in multilingual LLMs using same-language and cross-language conflicts across six languages, four domains (rule-following, safety, task-execution, persona), and three hierarchy types (system-user, system-tool, user-tool).","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-07-26","firstSeenAt":"2026-08-19","recognitionConfidence":0.95,"links":{"report":"https://arxiv.org/abs/2607.23545","pdf":"https://arxiv.org/pdf/2607.23545","project":null,"code":"https://github.com/g1moon/Language-Shapes-IH","data":null,"hfPaper":"https://huggingface.co/papers/2607.23545"},"evidence":{"snippet":"We introduce XIH-Bench, a benchmark for multilingual IH evaluation with both same-language and cross-language conflicts across six languages, four domains, and three IH settings.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":2,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.23545"},"ranking":{"90d":{"score":25,"rank":301,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates instruction hierarchy compliance in multilingual LLMs using same-language and cross-language conflicts across six languages, four domains (rule-following, safety, task-execution, persona), and three hierarchy types (system-user, system-tool, user-tool).","whyItMatters":"Existing instruction hierarchy benchmarks are largely English-centric, leaving a gap in assessing multilingual safety and reliability. This benchmark provides a reusable protocol to measure how language choice affects model compliance, supporting safer deployment in multilingual contexts.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"a550351d2dc3302ee2fca157c9db57278a19262f3fd0cf3b99c44e2e41ffcd8c"},"motivation":"Instruction hierarchy (IH) requires models to prioritize instructions by source, ensuring that higher-priority instructions override lower-priority ones.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"acceptance_claimed","venue":"EMNLP 2026 (Main)","evidence":"Accepted to EMNLP 2026 (Main). Code and data are available at https://github.com/g1moon/Language-Shapes-IH","evidenceUrl":"https://arxiv.org/abs/2607.23545","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-31T07:28:39.449760Z"},"venueAttempts":[{"venueName":"EMNLP 2026 (Main)","reviewStatus":"accepted","decisionRaw":"Accepted to EMNLP 2026 (Main). Code and data are available at https://github.com/g1moon/Language-Shapes-IH","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2607.23545","observedAt":"2026-08-31T07:28:39.449760Z","rawValue":"Accepted to EMNLP 2026 (Main). Code and data are available at https://github.com/g1moon/Language-Shapes-IH","level":"author-claim"}]}],"publishers":[{"name":"g1moon","organizationType":"community","sourceUrl":"https://github.com/g1moon/Language-Shapes-IH","role":"benchmark-publisher"}],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_xl-docbench_d4b9f759","familyId":"bmf_7ec798d59d5d","name":"XL-DocBench","oneLine":"XL-DocBench evaluates evidence-grounded long-document understanding with 1,519 human-verified questions from six professional domains, contexts up to 2,303 pages, multi-page evidence, and typed reasoning rules.","area":"Language & Knowledge","applicationDomains":["Health & Life Sciences","Finance & Economics"],"primaryDomain":"Health & Life Sciences","industrySectors":["Pharma & Biotech","Financial Services"],"capabilities":[],"topics":["cs.CL"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-21","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.00036","pdf":"https://arxiv.org/pdf/2608.00036","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2608.00036"},"evidence":{"snippet":"We introduce XL-DocBench, a fully human-verified benchmark for extra-long document understanding, with 1,519 retained questions from six professional domains and contexts up to 2,303 pages.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.00036"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"XL-DocBench evaluates evidence-grounded long-document understanding with 1,519 human-verified questions from six professional domains, contexts up to 2,303 pages, multi-page evidence, and typed reasoning rules.","whyItMatters":"Professional workflows require traceable answers from long documents; this benchmark fills a gap in multi-page and structured reasoning evaluation, enabling failure attribution.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"364074b787831af6ba8349ef106978e3cd713d305573340cd522c517978db1d5"},"motivation":"Real-world document tasks often ask professionals to answer questions from annual reports, regulations, clinical guidelines, and technical manuals that span hundreds or thousands of pages.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.00036","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"cross-domain"},{"id":"catalog_1b9f8b7eb2079c7b","familyId":"catalog_family_1b9f8b7eb2079c7b","name":"XLSum English","oneLine":"Large-scale multilingual abstractive summarization dataset comprising 1 million professionally annotated article-summary pairs from BBC, covering 44 languages. XL-Sum is highly abstractive, concise, and of high quality, designed to encourage research on multilingual abstractive summarization tasks.","description":"Large-scale multilingual abstractive summarization dataset comprising 1 million professionally annotated article-summary pairs from BBC, covering 44 languages. XL-Sum is highly abstractive, concise, and of high quality, designed to encourage research on multilingual abstractive summarization tasks.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Language","Summarization"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/xlsum-english","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_1b9f8b7eb2079c7b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/xlsum-english"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"xlsum-english","url":"https://llm-stats.com/benchmarks/xlsum-english","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["language","summarization"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"bm_xplainverse_db4c59ee","familyId":"bmf_b434c8c362ed","name":"XPlainVerse","oneLine":"XPlainVerse evaluates deepfake detection and explanation quality, pairing real images with forgeries from twelve models and providing technical and simplified explanations, with metrics EntityScore and EvidenceScore for reasoning fidelity.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Transform Existing","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-03","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.03562","pdf":"https://arxiv.org/pdf/2607.03562","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.03562"},"evidence":{"snippet":"To this end, we introduce XPlainVerse, a large-scale benchmark designed for joint deepfake detection and human-centered explanation.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.03562"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"XPlainVerse evaluates deepfake detection and explanation quality, pairing real images with forgeries from twelve models and providing technical and simplified explanations, with metrics EntityScore and EvidenceScore for reasoning fidelity.","whyItMatters":"Existing benchmarks focus on classification accuracy, not explanation grounding. XPlainVerse aims to measure whether explanations are grounded in actual manipulations, which is key for trustworthy deployable detection.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"f81da848a9b1ff7c7f80672b7085e5f505787ca6dd37da08bd6d297db90d1e10"},"motivation":"As deepfake detection models increasingly produce natural language explanations, their reasoning often remains weakly grounded in visual artifacts, limiting reliability and user trust.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.03562","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_xrepotest_04d9b86c","familyId":"bmf_2ad2aabdcaba","name":"XREPOTEST","oneLine":"Evaluates multilingual repository-level unit test generation across Rust, Go, Julia, PHP, and Ruby using containerized execution and context augmentation strategies.","area":"Code & Software","applicationDomains":[],"primaryDomain":"General AI","industrySectors":["Software & Cloud"],"capabilities":[],"topics":["cs.SE"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2026-08-26","firstSeenAt":"2026-08-27","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2608.25939","pdf":"https://arxiv.org/pdf/2608.25939","project":null,"code":"https://github.com/solis-team/XRepoTest","data":null,"hfPaper":null},"evidence":{"snippet":"We introduce XREPOTEST, a multilingual repository-level benchmark for unit test generation spanning five underexplored languages: Rust, Go, Julia, PHP, and Ruby.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":33,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.25939"},"ranking":{"30d":{"score":36,"rank":58,"coverage":0.85,"confidence":"High"},"90d":{"score":41,"rank":141,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates multilingual repository-level unit test generation across Rust, Go, Julia, PHP, and Ruby using containerized execution and context augmentation strategies.","whyItMatters":"Exposes the gap between standalone and repository-level test generation, providing metrics like invocation rate to measure whether generated tests exercise intended functionality.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:12:36.511123Z","inputHash":"d7bb169362cc6378b688a7f6d7236c7e86dbbaef546418fb523876255343dc15"},"motivation":"Large language models (LLMs) have shown promise for automated unit test generation, but existing evaluations largely rely on standalone settings and a narrow set of programming languages, overestimating real-world readiness.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:12:36.511123Z","model":"deepseek-v4-pro","decisionReason":"Provides a formally named benchmark with a public code repository and dataset on Hugging Face, a containerized evaluation framework, and multiple scoring metrics.","canonicalNameSource":"paper_title","canonicalNameEvidence":"XREPOTEST: Benchmarking Multilingual Repository-Level Unit Test Generation for Large Language Models"},"publication":{"status":"acceptance_claimed","venue":"EMNLP Main 2026","evidence":"Accepted to EMNLP Main 2026","evidenceUrl":"https://arxiv.org/abs/2608.25939","source":"arxiv-comments","evidenceLevel":"author-claim","verifiedAt":"2026-08-27T04:12:10.575570Z"},"venueAttempts":[{"venueName":"EMNLP Main 2026","reviewStatus":"accepted","decisionRaw":"Accepted to EMNLP Main 2026","evidence":[{"sourceType":"arxiv-comments","sourceUrl":"https://arxiv.org/abs/2608.25939","observedAt":"2026-08-27T04:12:10.575570Z","rawValue":"Accepted to EMNLP Main 2026","level":"author-claim"}]}],"attentionForecast":{"score":30,"confidence":"Low","horizon":"7d","reason":"A multilingual repository-level unit test benchmark with code and data release may attract moderate interest from software engineering and LLM evaluation communities."},"evaluationMode":"public_reusable","publishers":[{"name":"Solis Team","organizationType":"academic-lab","sourceUrl":"https://github.com/solis-team/XRepoTest","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Coding & Software Engineering"],"domainScope":"general"},{"id":"catalog_524e7cc577548477","familyId":"catalog_family_524e7cc577548477","name":"XSTest","oneLine":"XSTest is a test suite designed to identify exaggerated safety behaviours in large language models. It comprises 450 prompts: 250 safe prompts across ten prompt types that well-calibrated models should not refuse to comply with, and 200 unsafe prompts as contrasts that models should refuse. The benchmark systematically evaluates whether models refuse to respond to clearly safe prompts due to overly cautious safety mechanisms.","description":"XSTest is a test suite designed to identify exaggerated safety behaviours in large language models. It comprises 450 prompts: 250 safe prompts across ten prompt types that well-calibrated models should not refuse to comply with, and 200 unsafe prompts as contrasts that models should refuse. The benchmark systematically evaluates whether models refuse to respond to clearly safe prompts due to overly cautious safety mechanisms.","area":"Safety & Trustworthiness","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Safety"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/xstest","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_524e7cc577548477"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/xstest"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"xstest","url":"https://llm-stats.com/benchmarks/xstest","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["safety"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Safety & Trustworthiness"],"domainScope":"general"},{"id":"catalog_e2642e3ad267e6f8","familyId":"catalog_family_e2642e3ad267e6f8","name":"YC-Bench","oneLine":"YC-Bench evaluates agents on long-horizon, open-ended business and investment decision-making. The reported metric is the final assets (fund value, in US dollars) accumulated by the agent over the course of the simulation.","description":"YC-Bench evaluates agents on long-horizon, open-ended business and investment decision-making. The reported metric is the final assets (fund value, in US dollars) accumulated by the agent over the course of the simulation.","area":"Agents & Tool Use","applicationDomains":["Finance & Economics"],"primaryDomain":"Finance & Economics","industrySectors":[],"capabilities":[],"topics":["Finance","Agents"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/yc-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_e2642e3ad267e6f8"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/yc-bench"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"yc-bench","url":"https://llm-stats.com/benchmarks/yc-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["finance","agents"],"catalogModelCount":1,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"specific"},{"id":"bm_yocausal_0e0be7af","familyId":"bmf_8035539671b8","name":"YoCausal","oneLine":"YoCausal is a two-level benchmark that evaluates video diffusion models' understanding of temporal causality using the Violation of Expectation paradigm. It temporally reverses real-world videos as counterfactual samples and introduces two metrics: the Reverse Surprise Index (RSI) for arrow-of-time perception and the Causality Cognition Index (CCI) for disentangling genuine causal reasoning from temporal bias.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-05-28","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2605.30346","pdf":"https://arxiv.org/pdf/2605.30346","project":"https://www.youzhexie.me/papers/YoCausal/index.html","code":"https://github.com/youzhe0305/YoCausal","data":null,"hfPaper":"https://huggingface.co/papers/2605.30346"},"evidence":{"snippet":"We present YoCausal, a two-level benchmark inspired by the Violation of Expectation (VoE) paradigm from cognitive science.","reasonCodes":["exact coined title identity tied to benchmark evidence","benchmark release stated in one sentence","evaluation protocol evidence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":56,"hfDailySubmittedAt":null,"githubStars":36,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2605.30346"},"ranking":{},"description":"YoCausal is a two-level benchmark that evaluates video diffusion models' understanding of temporal causality using the Violation of Expectation paradigm. It temporally reverses real-world videos as counterfactual samples and introduces two metrics: the Reverse Surprise Index (RSI) for arrow-of-time perception and the Causality Cognition Index (CCI) for disentangling genuine causal reasoning from temporal bias.","whyItMatters":"Existing video benchmarks rely on synthetic data and do not separate temporal-direction awareness from true causal understanding. YoCausal provides a protocol that isolates causal cognition from temporal bias, enabling evaluation of whether models genuinely understand cause-and-effect relationships or only statistical patterns.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"c28262a53045ac1b371f3ee4cf2456813b0634cd54fc48006be0075aab8cde90"},"motivation":"As video diffusion models (VDMs) advance toward world models, a key question arises: do they truly understand causality, or merely overfit to statistical temporal patterns?","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2605.30346","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"publishers":[{"name":"National Yang Ming Chiao Tung University","organizationType":"academic-lab","sourceUrl":"https://github.com/youzhe0305/YoCausal","role":"benchmark-publisher"},{"name":"Shanda AI Research Tokyo","organizationType":"company-research-lab","sourceUrl":"https://github.com/youzhe0305/YoCausal","role":"benchmark-publisher"}],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_yomi-bench_e60b9cad","familyId":"bmf_3edc7fa51add","name":"YOMI-Bench","oneLine":"YOMI-Bench evaluates kanji reading and phonological understanding in LLMs for Japanese through four tasks. It is used in a study assessing multilingual and Japanese-specific models.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CL"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"2026-07-01","firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2607.00664","pdf":"https://arxiv.org/pdf/2607.00664","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2607.00664"},"evidence":{"snippet":"We propose YOMI-Bench, a benchmark for evaluating kanji reading and phonological understanding of large language models (LLMs) for Japanese.","reasonCodes":["coined title prefix ending in Bench or Benchmark","benchmark release stated in one sentence","evaluation protocol evidence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":null,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2607.00664"},"ranking":{"90d":{"score":null,"rank":null,"coverage":0.0,"confidence":"Low"}},"description":"YOMI-Bench evaluates kanji reading and phonological understanding in LLMs for Japanese through four tasks. It is used in a study assessing multilingual and Japanese-specific models.","whyItMatters":"There is no standalone public comparison path or shared artifact; the benchmark primarily supports the paper's finding that LLMs struggle with kanji reading.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"045ed5471782aa578c07d613417991851b229fd20a0c412e9aeae046a55fb502"},"motivation":"We propose YOMI-Bench, a benchmark for evaluating kanji reading and phonological understanding of large language models (LLMs) for Japanese.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2607.00664","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_5e2260ba16607194","familyId":"catalog_family_5e2260ba16607194","name":"ZClawBench","oneLine":"ZClawBench evaluates Claw-style agent task execution quality, measuring a model's ability to autonomously complete complex multi-step coding tasks in real-world environments.","description":"ZClawBench evaluates Claw-style agent task execution quality, measuring a model's ability to autonomously complete complex multi-step coding tasks in real-world environments.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic","Agents","Code"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://docs.z.ai/guides/llm/glm-5-turbo","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_5e2260ba16607194"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/zclawbench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/zclawbench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"zClawBench","url":"https://benchlm.ai/benchmarks/zclawbench","paperUrl":"https://docs.z.ai/guides/llm/glm-5-turbo","year":"2026","fullName":"ZClawBench","format":"End-to-end agent benchmark","tasks":"OpenClaw agent workflows","successorKey":null},{"catalog":"llm-stats","sourceId":"zclawbench","url":"https://llm-stats.com/benchmarks/zclawbench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","agents","code"],"catalogModelCount":4,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_39277112cb15425b","familyId":"catalog_family_39277112cb15425b","name":"ZebraLogic","oneLine":"ZebraLogic is an evaluation framework for assessing large language models' logical reasoning capabilities through logic grid puzzles derived from constraint satisfaction problems (CSPs). The benchmark consists of 1,000 programmatically generated puzzles with controllable and quantifiable complexity, revealing a 'curse of complexity' where model accuracy declines significantly as problem complexity grows.","description":"ZebraLogic is an evaluation framework for assessing large language models' logical reasoning capabilities through logic grid puzzles derived from constraint satisfaction problems (CSPs). The benchmark consists of 1,000 programmatically generated puzzles with controllable and quantifiable complexity, revealing a 'curse of complexity' where model accuracy declines significantly as problem complexity grows.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Reasoning"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/zebralogic","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_39277112cb15425b"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/zebralogic"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"zebralogic","url":"https://llm-stats.com/benchmarks/zebralogic","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["reasoning"],"catalogModelCount":8,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_702c994cae574a63","familyId":"catalog_family_702c994cae574a63","name":"ZeroBench","oneLine":"ZEROBench is a challenging vision benchmark designed to test models on zero-shot visual understanding tasks.","description":"ZEROBench is a challenging vision benchmark designed to test models on zero-shot visual understanding tasks.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded","Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_702c994cae574a63"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/zerobench"},{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/zerobench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"zeroBench","url":"https://benchlm.ai/benchmarks/zerobench","paperUrl":"https://ai.meta.com/static-resource/muse-spark-eval-methodology","year":"2026","fullName":"ZeroBench","format":"Multi-step visual reasoning","tasks":"100 visual reasoning questions","successorKey":null},{"catalog":"llm-stats","sourceId":"zerobench","url":"https://llm-stats.com/benchmarks/zerobench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodalGrounded","multimodal","reasoning","vision"],"catalogModelCount":10,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"catalog_f5593566c5676379","familyId":"catalog_family_f5593566c5676379","name":"ZeroBench w/ Python","oneLine":"A Python-assisted ZeroBench_main evaluation reported as pass@5.","description":"A Python-assisted ZeroBench_main evaluation reported as pass@5.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodalgrounded"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://www.kimi.com/blog/kimi-k3","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_f5593566c5676379"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/zerobenchpython"}],"catalogSources":[{"catalog":"benchlm","sourceId":"zeroBenchPython","url":"https://benchlm.ai/benchmarks/zerobenchpython","paperUrl":"https://www.kimi.com/blog/kimi-k3","year":"2026","fullName":"ZeroBench_main with Python","format":"Pass@5","tasks":"Visual reasoning questions with Python","successorKey":null}],"catalogCategories":["multimodalGrounded"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_c670958d92dab5d2","familyId":"catalog_family_c670958d92dab5d2","name":"ZEROBench-Sub","oneLine":"ZEROBench-Sub is a subset of the ZEROBench benchmark.","description":"ZEROBench-Sub is a subset of the ZEROBench benchmark.","area":"Multimodal","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Multimodal","Reasoning","Vision"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://llm-stats.com/benchmarks/zerobench-sub","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c670958d92dab5d2"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"llm-stats","url":"https://llm-stats.com/benchmarks/zerobench-sub"}],"catalogSources":[{"catalog":"llm-stats","sourceId":"zerobench-sub","url":"https://llm-stats.com/benchmarks/zerobench-sub","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["multimodal","reasoning","vision"],"catalogModelCount":5,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_zipbench_2610de18","familyId":"bmf_71f839953810","name":"ZIPBench","oneLine":"ZIPBench is a zero-shot personalization benchmark with 1.5K users, graph-mined personas, and 40K generated images for evaluating text-to-image personalization.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.AI"],"construction":"Aggregate Existing","annotation":"Mixed","readiness":"Paper only","releasedAt":"2026-06-07","firstSeenAt":"2026-08-19","recognitionConfidence":0.85,"links":{"report":"https://arxiv.org/abs/2606.08841","pdf":"https://arxiv.org/pdf/2606.08841","project":null,"code":null,"data":null,"hfPaper":"https://huggingface.co/papers/2606.08841"},"evidence":{"snippet":"We introduce ZIPBench, the first zero-shot personalization benchmark with 1.5K users, graph-mined personas, and 40K generated images.","reasonCodes":["exact named benchmark artifact released in abstract","benchmark release stated in one sentence"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":1,"hfDailySubmittedAt":null,"githubStars":null,"githubScope":null,"hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2606.08841"},"ranking":{"90d":{"score":44,"rank":null,"coverage":0.15,"confidence":"Low"}},"description":"ZIPBench is a zero-shot personalization benchmark with 1.5K users, graph-mined personas, and 40K generated images for evaluating text-to-image personalization.","whyItMatters":"It addresses the need for evaluating personalization without user-specific data, assessing alignment with individual aesthetic preferences.","copyGeneration":{"model":"deepseek-v4-flash","generatedAt":"2026-08-20T15:22:10.044731Z","inputHash":"52cf7f930ff0f25beae5c29e882e9e91d63d983a51672d0c940dc713beb51f67"},"motivation":"Text-to-image diffusion models are increasingly deployed in open-ended creative contexts, yet their outputs remain impersonal, optimized for aggregate aesthetics rather than individual taste.","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2606.08841","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-31T07:28:39.449760Z"},"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"bm_pi-sub-a-physics-informed-synthetic-underw_d13ec116","familyId":"bmf_c2c5f042ab2d","name":"π-SUB","oneLine":"Evaluates underwater image enhancement models using a synthetic paired dataset generated by a physics-informed framework with depth-dependent irradiance and Jerlov water types.","area":"Vision & 3D","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["cs.CV"],"construction":"Original Synthetic","annotation":"Machine Generated","readiness":"Runnable","releasedAt":"2026-08-11","firstSeenAt":"2026-08-19","recognitionConfidence":0.55,"links":{"report":"https://arxiv.org/abs/2608.10589","pdf":"https://arxiv.org/pdf/2608.10589","project":null,"code":"https://github.com/airl-iisc/pi-SUB","data":null,"hfPaper":null},"evidence":{"snippet":"This paper presents $\\pi$-SUB, a physics-informed framework for generating synthetic underwater benchmark datasets that bridges the synthetic-to-real gap for Underwater Image Enhancement (UIE).","reasonCodes":["benchmark term in abstract","benchmark release stated in one sentence","public artifact URL"]},"dataStatus":"primary-source-indexed","demo":false,"attention":{"asOf":"2026-08-31","hfPaperUpvotes":0,"hfDailySubmittedAt":null,"githubStars":0,"githubScope":"benchmark_repo","hfDatasetDownloads":null,"hfDatasetLikes":null},"source":{"type":"arxiv","id":"2608.10589"},"ranking":{"30d":{"score":8,"rank":166,"coverage":0.85,"confidence":"High"},"90d":{"score":15,"rank":398,"coverage":0.7,"confidence":"Medium"}},"description":"Evaluates underwater image enhancement models using a synthetic paired dataset generated by a physics-informed framework with depth-dependent irradiance and Jerlov water types.","whyItMatters":"Offers a hyper-realistic and generalizable synthetic benchmark to close the synthetic-to-real gap for underwater image enhancement.","copyGeneration":{"model":"deepseek-v4-pro","generatedAt":"2026-08-27T04:14:52.836173Z","inputHash":"061c934c33b1a95e07d374b61b06c57f3ccea1b538dd8a7eae2c297c43adfb28"},"motivation":"This paper presents $\\pi$-SUB, a physics-informed framework for generating synthetic underwater benchmark datasets that bridges the synthetic-to-real gap for Underwater Image Enhancement (UIE).","constructionDetail":"Unknown — the indexer does not infer construction details without explicit source evidence.","curation":{"state":"ai-reviewed","reviewedAt":"2026-08-27T04:14:52.836173Z","model":"deepseek-v4-pro","decisionReason":"The dataset and code are publicly available on GitHub, and the paper provides quantitative evaluation protocols comparing models on real benchmarks.","canonicalNameSource":"abstract","canonicalNameEvidence":"This paper presents $\\pi$-SUB, a physics-informed framework for generating synthetic underwater benchmark datasets"},"publication":{"status":"unverified","venue":null,"evidence":null,"evidenceUrl":"https://arxiv.org/abs/2608.10589","source":"arxiv-metadata","evidenceLevel":"unverified","verifiedAt":"2026-08-19T11:05:13.395391Z"},"attentionForecast":{"score":44,"confidence":"Medium","horizon":"7d","reason":"The benchmark focuses on a specialized underwater image enhancement niche, with public code and dataset available, likely to draw modest attention from researchers in that area."},"evaluationMode":"public_reusable","publishers":[{"name":"AIRL, Indian Institute of Science","organizationType":"academic-lab","sourceUrl":"https://github.com/airl-iisc/pi-SUB","role":"benchmark-publisher"}],"displayEligible":true,"capabilityGroups":["Multimodal Perception"],"domainScope":"general"},{"id":"lib_tau_bench","familyId":"family_tau_bench","name":"τ-bench","oneLine":"Established benchmark family · Agents & Tool Use.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents & Tool Use"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2024-01-01","releaseDatePrecision":"year","firstRelease":{"year":2024,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2406.12045","pdf":null,"project":"https://github.com/sierra-research/tau-bench","code":"https://github.com/sierra-research/tau-bench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_tau_bench"},"ranking":{},"recordType":"family","aliases":["tau-bench","TAU-bench"],"sourceAttribution":[{"role":"official-repository","url":"https://github.com/sierra-research/tau-bench"}],"adoptionRefs":["openai-gpt5","anthropic-claude4"],"modelReportReferences":[{"sourceId":"openai-gpt5","url":"https://openai.com/index/introducing-gpt-5/","provider":"OpenAI","model":"GPT-5"},{"sourceId":"anthropic-claude4","url":"https://www.anthropic.com/news/claude-4","provider":"Anthropic","model":"Claude 4"}],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"capabilityGroups":["Agents"],"domainScope":"general","catalogSources":[{"catalog":"benchlm","sourceId":"tauBench","url":"https://benchlm.ai/benchmarks/tau-bench","paperUrl":"https://arxiv.org/abs/2406.12045","year":"2024","fullName":"Tool-Agent-User Benchmark","format":"Domain-specific pass^1 through pass^4 task success","tasks":"Airline and retail task sets in the archived 2024 release","successorKey":null},{"catalog":"llm-stats","sourceId":"tau-bench","url":"https://llm-stats.com/benchmarks/tau-bench","datasetSlug":null,"versionCount":null,"subsetCount":null,"rowCount":null,"updatedAt":null,"community":false}],"catalogCategories":["agentic","reasoning","general","agents","tool calling"],"catalogModelCount":6,"catalogStarCount":0},{"id":"lib_tau2_bench","familyId":"family_tau_bench","name":"τ²-bench","oneLine":"Established benchmark variant · Agents & Tool Use.","area":"Agents & Tool Use","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agents & Tool Use"],"construction":"Unknown","annotation":"Unknown","readiness":"Runnable","releasedAt":"2025-01-01","releaseDatePrecision":"year","firstRelease":{"year":2025,"date":null},"firstSeenAt":"2026-08-19","recognitionConfidence":1.0,"links":{"report":"https://arxiv.org/abs/2506.07982","pdf":null,"project":"https://github.com/sierra-research/tau2-bench","code":"https://github.com/sierra-research/tau2-bench","data":null,"hfPaper":null},"evidence":{"snippet":"Reviewed Library record; follow the linked benchmark source for its definition.","reasonCodes":["editorial Library seed","source attribution retained"]},"dataStatus":"editorial-source-reviewed","demo":false,"attention":{},"source":{"type":"library","id":"lib_tau2_bench"},"ranking":{},"recordType":"variant","aliases":["tau2-bench","TAU2-bench"],"sourceAttribution":[{"role":"official-repository","url":"https://github.com/sierra-research/tau2-bench"}],"adoptionRefs":[],"modelReportReferences":[],"catalogDiscoveryRefs":["llm-stats","benchlm"],"catalogDiscoverySources":[{"sourceId":"llm-stats","url":"https://llm-stats.com/benchmarks"},{"sourceId":"benchlm","url":"https://benchlm.ai/benchmarks"}],"usageObservations":[],"variantOf":"lib_tau_bench","capabilityGroups":["Agents"],"domainScope":"general"},{"id":"catalog_c72fe98e0d4560ef","familyId":"catalog_family_c72fe98e0d4560ef","name":"τ²-bench Airline","oneLine":"τ²-bench Airline tests conversational agents on airline customer-service tasks governed by domain policy and database-changing tools.","description":"τ²-bench Airline tests conversational agents on airline customer-service tasks governed by domain policy and database-changing tools.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2506.07982","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_c72fe98e0d4560ef"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/tau2airline"}],"catalogSources":[{"catalog":"benchlm","sourceId":"tau2Airline","url":"https://benchlm.ai/benchmarks/tau2airline","paperUrl":"https://arxiv.org/abs/2506.07982","year":"2025","fullName":"τ²-Bench Airline Domain","format":"Domain success under a published trial policy","tasks":"Airline customer-service tasks","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_a5f19cb5cb098ef9","familyId":"catalog_family_a5f19cb5cb098ef9","name":"τ²-bench results","oneLine":"This route is a sourced ledger for published τ²-bench results. Most current rows come from Artificial Analysis's telecom implementation, while named provider rows can use telecom, airline, retail, or aggregate setups.","description":"This route is a sourced ledger for published τ²-bench results. Most current rows come from Artificial Analysis's telecom implementation, while named provider rows can use telecom, airline, retail, or aggregate setups.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://arxiv.org/abs/2506.07982","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_a5f19cb5cb098ef9"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/tau2-bench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"tau2Bench","url":"https://benchlm.ai/benchmarks/tau2-bench","paperUrl":"https://arxiv.org/abs/2506.07982","year":"2025","fullName":"τ²-Bench Tool-Agent-User Evaluation","format":"Published domain success or pass^k results","tasks":"Airline, retail, and telecom customer-service task sets","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"},{"id":"catalog_fb6c7f63dd4e6a34","familyId":"catalog_family_fb6c7f63dd4e6a34","name":"τ³-bench results","oneLine":"τ³-bench is the current evolution of Sierra's tool-agent-user framework, adding corrected task releases and newer knowledge and voice evaluation modes alongside airline, retail, and telecom.","description":"τ³-bench is the current evolution of Sierra's tool-agent-user framework, adding corrected task releases and newer knowledge and voice evaluation modes alongside airline, retail, and telecom.","area":"Language & Knowledge","applicationDomains":[],"primaryDomain":"General AI","industrySectors":[],"capabilities":[],"topics":["Agentic"],"construction":"Unknown","annotation":"Unknown","readiness":"Paper only","releasedAt":"0001-01-01","releaseDatePrecision":"unknown","firstRelease":{"year":null,"date":null},"firstSeenAt":"2026-08-27","recognitionConfidence":0.5,"links":{"report":"https://github.com/sierra-research/tau2-bench","pdf":null,"project":null,"code":null,"data":null,"hfPaper":null},"evidence":{"snippet":"Listed by BenchLM or llm-stats; original-source verification is pending.","reasonCodes":["external catalog listing"]},"dataStatus":"catalog-listed-unverified","demo":false,"attention":{},"source":{"type":"catalog","id":"catalog_fb6c7f63dd4e6a34"},"ranking":{},"recordType":"catalog-entry","aliases":[],"sourceAttribution":[{"role":"benchlm","url":"https://benchlm.ai/benchmarks/tau3-bench"}],"catalogSources":[{"catalog":"benchlm","sourceId":"tau3Bench","url":"https://benchlm.ai/benchmarks/tau3-bench","paperUrl":"https://github.com/sierra-research/tau2-bench","year":"2026","fullName":"τ³-Bench Tool-Agent-User Evaluation","format":"Published domain or average success results","tasks":"Corrected customer-service tasks plus knowledge and voice evaluation modes","successorKey":null}],"catalogCategories":["agentic"],"catalogModelCount":0,"catalogStarCount":0,"modelReportReferences":[],"usageObservations":[],"capabilityGroups":["Knowledge & Reasoning"],"domainScope":"general"}]}
