{
 "active_ids": [
  "aa_briefcase",
  "aa_lcr",
  "automationbench_aa",
  "critpt",
  "gdp_pdf_aa",
  "gdpval_aa",
  "scicode"
 ],
 "benchmarks": [
  {
   "aliases": [
    "AA Briefcase",
    "Artificial Analysis Briefcase"
   ],
   "canonical_id": "aa_briefcase",
   "category": "agentic",
   "disposition": "active",
   "id": "aa_briefcase",
   "models_covered": 0,
   "name": "AA-Briefcase",
   "reasons": [
    "qualifying current coverage"
   ],
   "summary": "Artificial Analysis's private 91-task agentic benchmark of long-horizon professional knowledge work, scored as combined Elo from rubric success, analytical quality, and presentation."
  },
  {
   "aliases": [
    "AA-LCR",
    "Artificial Analysis Long Context Reasoning"
   ],
   "canonical_id": "aa_lcr",
   "category": "long-context",
   "disposition": "active",
   "id": "aa_lcr",
   "models_covered": 0,
   "name": "AA-LCR (Artificial Analysis Long Context Reasoning)",
   "reasons": [
    "qualifying current coverage"
   ],
   "summary": "Artificial Analysis's 100-question test of whether a model can reason across ~100k-token real-world document sets, not just retrieve a stated fact."
  },
  {
   "aliases": [],
   "canonical_id": "abstention_bench",
   "category": "safety",
   "disposition": "unassessed",
   "id": "abstention_bench",
   "models_covered": 0,
   "name": "AbstentionBench",
   "reasons": [],
   "summary": "A FAIR/Meta benchmark of 20 aggregated datasets testing whether models decline to answer questions that cannot or should not be answered confidently, finding reasoning fine-tuning makes this worse."
  },
  {
   "aliases": [
    "ePiC (BIG-bench 100-proverb cut)",
    "abstract_narrative_understanding_4_distractors",
    "abstract_narrative_understanding_9_distractors",
    "abstract_narrative_understanding_99_distractors"
   ],
   "canonical_id": "abstract_narrative_understanding",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "abstract_narrative_understanding",
   "models_covered": 0,
   "name": "Abstract Narrative Understanding (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench task: pick the English proverb that matches a short crowdsourced story, with 4, 9 or 99 distractors."
  },
  {
   "aliases": [
    "BIG-bench ARC",
    "Chollet ARC (BIG-bench task)"
   ],
   "canonical_id": "abstraction_and_reasoning_corpus",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "abstraction_and_reasoning_corpus",
   "models_covered": 0,
   "name": "Abstraction and Reasoning Corpus (BIG-bench wrapping)",
   "reasons": [],
   "summary": "BIG-bench's text wrapping of Chollet's 400-task ARC evaluation set: infer a hidden grid rule from examples and emit the output grid as digit strings."
  },
  {
   "aliases": [
    "ACI-BENCH"
   ],
   "canonical_id": "aci_bench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "aci_bench",
   "models_covered": 0,
   "name": "ACI-Bench",
   "reasons": [],
   "summary": "Tests whether a model can turn a doctor-patient conversation transcript into a structured clinical note; 207 real dialogue-note pairs, the largest public dataset of its kind at publication."
  },
  {
   "aliases": [
    "Ancient Chinese Language Understanding Evaluation"
   ],
   "canonical_id": "aclue",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "aclue",
   "models_covered": 0,
   "name": "ACLUE (Ancient Chinese Language Understanding Evaluation)",
   "reasons": [],
   "summary": "Fifteen four-option ancient-Chinese tasks spanning lexicon, syntax, poetry, medicine and culture; 2023 zero-shot scores sat only a little above chance."
  },
  {
   "aliases": [
    "ACP Bench",
    "acp-bench"
   ],
   "canonical_id": "acp_bench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "acp_bench",
   "models_covered": 0,
   "name": "ACPBench",
   "reasons": [],
   "summary": "IBM's boolean- and multiple-choice test of seven atomic reasoning skills needed for planning -- action applicability, reachability, justification, landmarks and more -- across 13 formal domains."
  },
  {
   "aliases": [
    "ACPBench Hard"
   ],
   "canonical_id": "acp_bench_hard",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "acp_bench_hard",
   "models_covered": 0,
   "name": "ACPBench-Hard",
   "reasons": [],
   "summary": "IBM's harder, generative companion to ACPBench: the same planning-reasoning skills plus a new 'next action' task, answered as open text and checked by per-task validators, not multiple choice."
  },
  {
   "aliases": [
    "Adversarial GLUE",
    "AdvGLUE"
   ],
   "canonical_id": "adv_glue",
   "category": "composite",
   "disposition": "unassessed",
   "id": "adv_glue",
   "models_covered": 0,
   "name": "AdvGLUE (Adversarial GLUE)",
   "reasons": [],
   "summary": "Adversarial restatement of five GLUE tasks using word-level, sentence-level and human-written attacks; OpenCompass scores a public annotated development pack, not the original hidden test set."
  },
  {
   "aliases": [
    "Advanced Instruction Following"
   ],
   "canonical_id": "advancedif",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "advancedif",
   "models_covered": 0,
   "name": "AdvancedIF",
   "reasons": [],
   "summary": "Meta's 1,645-prompt instruction-following benchmark, expert-written and LLM-judged against per-prompt rubrics, covering complex single-turn instructions, multi-turn carried context and system-prompt steerability."
  },
  {
   "aliases": [],
   "canonical_id": "agent_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "agent_bench",
   "models_covered": 0,
   "name": "AgentBench",
   "reasons": [],
   "summary": "Tests LLMs as interactive agents across eight environments (OS, database, knowledge graph, card game, puzzles, household, web shopping, web browsing); most modern harnesses implement only its OS slice."
  },
  {
   "aliases": [],
   "canonical_id": "agent_memory_bench",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "agent_memory_bench",
   "models_covered": 0,
   "name": "agent-memory-bench",
   "reasons": [],
   "summary": "agent-memory-bench is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "agent_memory_benchmark_amb",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "agent_memory_benchmark_amb",
   "models_covered": 0,
   "name": "Agent Memory Benchmark (AMB)",
   "reasons": [],
   "summary": "Agent Memory Benchmark (AMB) is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "agent_memory_leaderboard",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "agent_memory_leaderboard",
   "models_covered": 0,
   "name": "Agent Memory Leaderboard",
   "reasons": [],
   "summary": "Agent Memory Leaderboard is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "agent_threat_bench",
   "category": "safety",
   "disposition": "unassessed",
   "id": "agent_threat_bench",
   "models_covered": 0,
   "name": "AgentThreatBench",
   "reasons": [],
   "summary": "Small Inspect AI suite that scores whether an LLM agent completes its task and separately whether it resists prompt-injection attacks drawn from the OWASP Agentic Top 10."
  },
  {
   "aliases": [
    "AgentDojo"
   ],
   "canonical_id": "agentdojo",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "agentdojo",
   "models_covered": 0,
   "name": "AgentDojo",
   "reasons": [],
   "summary": "Stateful tool-using agent suites that score both task utility and resistance to prompt injections planted in tool outputs."
  },
  {
   "aliases": [
    "AgentHarm"
   ],
   "canonical_id": "agentharm",
   "category": "safety",
   "disposition": "unassessed",
   "id": "agentharm",
   "models_covered": 0,
   "name": "AgentHarm",
   "reasons": [],
   "summary": "Graded multi-step tool-use tasks that measure whether an agent will carry out harmful requests such as fraud or cybercrime, with a matched benign control set."
  },
  {
   "aliases": [
    "Agentic Misalignment",
    "Anthropic agentic misalignment"
   ],
   "canonical_id": "agentic_misalignment",
   "category": "safety",
   "disposition": "unassessed",
   "id": "agentic_misalignment",
   "models_covered": 0,
   "name": "Agentic misalignment",
   "reasons": [],
   "summary": "Fictional corporate-agent scenarios that test whether a model blackmails, leaks, or otherwise acts as an insider when it faces replacement or a goal conflict."
  },
  {
   "aliases": [],
   "canonical_id": "agieval",
   "category": "composite",
   "disposition": "unassessed",
   "id": "agieval",
   "models_covered": 0,
   "name": "AGIEval",
   "reasons": [],
   "summary": "8,062 real questions from 20 human standardized exams -- Gaokao, SAT, LSAT, a Chinese bar exam, civil-service logic tests and math competitions -- scored mostly by multiple-choice accuracy."
  },
  {
   "aliases": [
    "AI2 Diagrams",
    "AI2D-Test"
   ],
   "canonical_id": "ai2d",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "ai2d",
   "models_covered": 39,
   "name": "AI2D (AI2 Diagrams)",
   "reasons": [],
   "summary": "Multiple-choice question answering over labeled grade-school science diagrams, testing whether a model can connect diagram text, structure and layout to a question."
  },
  {
   "aliases": [],
   "canonical_id": "ai_agent_benchmark",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "ai_agent_benchmark",
   "models_covered": 0,
   "name": "ai-agent-benchmark",
   "reasons": [],
   "summary": "ai-agent-benchmark is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "ai_agent_benchmark_leaderboards",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "ai_agent_benchmark_leaderboards",
   "models_covered": 0,
   "name": "AI Agent Benchmark Leaderboards",
   "reasons": [],
   "summary": "AI Agent Benchmark Leaderboards is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "ai_agent_benchmark_results",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "ai_agent_benchmark_results",
   "models_covered": 0,
   "name": "AI Agent Benchmark Results",
   "reasons": [],
   "summary": "AI Agent Benchmark Results is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [
    "Aider's polyglot benchmark",
    "polyglot-benchmark"
   ],
   "canonical_id": "aider_polyglot",
   "category": "coding",
   "disposition": "unassessed",
   "id": "aider_polyglot",
   "models_covered": 116,
   "name": "Aider Polyglot Benchmark",
   "reasons": [],
   "summary": "225 hard Exercism exercises across six languages, scoring whether a model can write and then fix its own code from failing test output."
  },
  {
   "aliases": [
    "American Invitational Mathematics Examination",
    "AIME 1983-2024"
   ],
   "canonical_id": "aime",
   "category": "math",
   "disposition": "unassessed",
   "id": "aime",
   "models_covered": 0,
   "name": "AIME (American Invitational Mathematics Examination)",
   "reasons": [],
   "summary": "The American Invitational Mathematics Examination as an LLM test: exact integer answers to contest problems, packaged as a 1983\u20132024 historical set and as yearly 30-problem sittings."
  },
  {
   "aliases": [
    "AIME24",
    "AIME 2024 I and II"
   ],
   "canonical_id": "aime_2024",
   "category": "math",
   "disposition": "unassessed",
   "id": "aime_2024",
   "models_covered": 0,
   "name": "AIME 2024",
   "reasons": [],
   "summary": "The 30 problems from the 2024 American Invitational Mathematics Examination, an exact-answer competition-math test now well past its useful ceiling for frontier models."
  },
  {
   "aliases": [
    "AIME25",
    "AIME 2025 I and II"
   ],
   "canonical_id": "aime_2025",
   "category": "math",
   "disposition": "unassessed",
   "id": "aime_2025",
   "models_covered": 29,
   "name": "AIME 2025",
   "reasons": [],
   "summary": "The 30 problems from the 2025 American Invitational Mathematics Examination, scored as an exact-answer test of competition-level math reasoning."
  },
  {
   "aliases": [
    "AIME26",
    "AIME 2026 I and II"
   ],
   "canonical_id": "aime_2026",
   "category": "math",
   "disposition": "unassessed",
   "id": "aime_2026",
   "models_covered": 0,
   "name": "AIME 2026",
   "reasons": [],
   "summary": "The 30 problems from the 2026 American Invitational Mathematics Examination, administered in February 2026, scored as an exact-answer test of competition math reasoning."
  },
  {
   "aliases": [
    "AIR-Bench",
    "AIRBench 2024",
    "AI Risk Benchmark"
   ],
   "canonical_id": "air_bench",
   "category": "safety",
   "disposition": "unassessed",
   "id": "air_bench",
   "models_covered": 0,
   "name": "AIR-Bench 2024",
   "reasons": [],
   "summary": "5,694 prompts judged against a 314-category safety taxonomy built from real government regulations and company policies; distinct from two other, unrelated benchmarks also named AIR-Bench."
  },
  {
   "aliases": [],
   "canonical_id": "air_bench_2024",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "air_bench_2024",
   "models_covered": 0,
   "name": "AIR-Bench 2024",
   "reasons": [],
   "summary": "AIR-Bench 2024 is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "air_bench_dataset",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "air_bench_dataset",
   "models_covered": 0,
   "name": "AIR-Bench-Dataset",
   "reasons": [],
   "summary": "AIR-Bench-Dataset is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "air_bench_live",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "air_bench_live",
   "models_covered": 0,
   "name": "AIR-BENCH Live",
   "reasons": [],
   "summary": "AIR-BENCH Live is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [
    "AlGhafa Evaluation Benchmark",
    "AlGhafa-Arabic-LLM-Benchmark-Native",
    "arabic_leaderboard_alghafa"
   ],
   "canonical_id": "alghafa",
   "category": "composite",
   "disposition": "unassessed",
   "id": "alghafa",
   "models_covered": 0,
   "name": "AlGhafa",
   "reasons": [],
   "summary": "TII's Arabic multiple-choice suite; HELM and OALL score nine native Hugging Face configs with public answers, not the translated COPA/OpenBookQA extras."
  },
  {
   "aliases": [
    "AlpacaEval 2.0",
    "Length-Controlled AlpacaEval",
    "LC AlpacaEval"
   ],
   "canonical_id": "alpaca_eval",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "alpaca_eval",
   "models_covered": 56,
   "name": "AlpacaEval",
   "reasons": [],
   "summary": "An automatic, LLM-judged win-rate test of instruction-following that is built and validated to track human preference votes."
  },
  {
   "aliases": [
    "OALL/ALRAGE"
   ],
   "canonical_id": "alrage",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "alrage",
   "models_covered": 0,
   "name": "ALRAGE",
   "reasons": [],
   "summary": "OALL's 2,106-item Arabic passage QA set; HELM grades free-form answers with GPT-4o, using the public train split as the test set."
  },
  {
   "aliases": [
    "Identifying Anachronisms"
   ],
   "canonical_id": "anachronisms",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "anachronisms",
   "models_covered": 0,
   "name": "Anachronisms",
   "reasons": [],
   "summary": "A 230-item BIG-bench Yes/No task that asks whether an English sentence contains a temporally impossible mix of people, objects, or dates."
  },
  {
   "aliases": [],
   "canonical_id": "analogical_similarity",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "analogical_similarity",
   "models_covered": 0,
   "name": "Analogical Similarity",
   "reasons": [],
   "summary": "A 323-item BIG-bench task that classifies how two short event sentences share objects, relations, and structure, using Gentner-style similarity labels."
  },
  {
   "aliases": [],
   "canonical_id": "analytic_entailment",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "analytic_entailment",
   "models_covered": 0,
   "name": "Analytic Entailment",
   "reasons": [],
   "summary": "A 70-item BIG-bench task that asks whether a second English sentence follows from the first by meaning alone, not by world facts."
  },
  {
   "aliases": [
    "AHB",
    "Animal Harm Benchmark",
    "Animal Norms In Moral Assessment",
    "inspect_evals/anima"
   ],
   "canonical_id": "anima",
   "category": "safety",
   "disposition": "unassessed",
   "id": "anima",
   "models_covered": 0,
   "name": "ANIMA (Animal Norms In Moral Assessment)",
   "reasons": [],
   "summary": "Inspect eval of animal-welfare moral reasoning across 13 dimensions; the public set is 115 questions, up from the paper's original 26."
  },
  {
   "aliases": [
    "Adversarial NLI"
   ],
   "canonical_id": "anli",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "anli",
   "models_covered": 0,
   "name": "ANLI (Adversarial NLI)",
   "reasons": [],
   "summary": "A three-round adversarial natural language inference benchmark where annotators iteratively wrote examples to fool the strongest model trained on all prior rounds."
  },
  {
   "aliases": [
    "HH-RLHF",
    "Anthropic RLHF dataset",
    "anthropic_hh_rlhf:subset=hh",
    "anthropic_hh_rlhf:subset=red_team"
   ],
   "canonical_id": "anthropic_hh_rlhf",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "anthropic_hh_rlhf",
   "models_covered": 0,
   "name": "Anthropic HH-RLHF (HELM Instruct)",
   "reasons": [],
   "summary": "HELM Instruct scores first human utterances from Anthropic HH-RLHF with a 1-5 Helpfulness critique; it does not train on or rank the chosen/rejected pairs."
  },
  {
   "aliases": [
    "AnthropicRedTeam",
    "anthropic-red-team",
    "hh-rlhf red-team-attempts"
   ],
   "canonical_id": "anthropic_red_team",
   "category": "safety",
   "disposition": "unassessed",
   "id": "anthropic_red_team",
   "models_covered": 0,
   "name": "Anthropic Red Team (HELM)",
   "reasons": [],
   "summary": "HELM safety scenario that scores a model's first reply to 38,961 Anthropic red-team openings with a 0\u20131 LLM-judge safety_score."
  },
  {
   "aliases": [
    "Attempt to Persuade Eval",
    "ape_eval",
    "AttemptPersuadeEval"
   ],
   "canonical_id": "ape",
   "category": "safety",
   "disposition": "unassessed",
   "id": "ape",
   "models_covered": 0,
   "name": "APE (Attempt to Persuade Eval)",
   "reasons": [],
   "summary": "Multi-turn eval of whether a model tries to persuade a simulated user; the headline is turn-1 attempt rate on harmful topics, not persuasion success."
  },
  {
   "aliases": [],
   "canonical_id": "apps",
   "category": "coding",
   "disposition": "unassessed",
   "id": "apps",
   "models_covered": 0,
   "name": "APPS (Automated Programming Progress Standard)",
   "reasons": [],
   "summary": "10,000 Python coding problems scraped from competitive-programming and interview sites across three difficulty tiers, graded by executing generated code against held-out test cases."
  },
  {
   "aliases": [
    "Arab Culture",
    "arab_culture",
    "ArabicCulture",
    "MBZUAI/ArabCulture"
   ],
   "canonical_id": "arabculture",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "arabculture",
   "models_covered": 0,
   "name": "ArabCulture",
   "reasons": [],
   "summary": "3,482 native Modern Standard Arabic three-choice commonsense items across 13 countries; lm-eval group arab_culture, not translated English CSQA."
  },
  {
   "aliases": [
    "Article Generation",
    "arabic enterprise content_generation"
   ],
   "canonical_id": "arabic_content_generation",
   "category": "generation",
   "disposition": "unassessed",
   "id": "arabic_content_generation",
   "models_covered": 0,
   "name": "Arabic Content Generation (HELM Arabic Enterprise)",
   "reasons": [],
   "summary": "HELM Arabic Enterprise task: write a Modern Standard Arabic business article from given facts and style; an LLM judge scores faithfulness, completeness and style."
  },
  {
   "aliases": [
    "aexams",
    "AEXAMS",
    "EXAMS (Arabic)"
   ],
   "canonical_id": "arabic_exams",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "arabic_exams",
   "models_covered": 0,
   "name": "Arabic EXAMS",
   "reasons": [],
   "summary": "Multiple-choice Arabic high-school exam questions across five subjects, run under two different harness names (arabic_exams, aexams) over the same 562-item set."
  },
  {
   "aliases": [
    "arabic_finance_mcq",
    "arabic_finance_bool",
    "arabic_finance_calculation",
    "Arabic Enterprise finance"
   ],
   "canonical_id": "arabic_finance",
   "category": "domain",
   "disposition": "unassessed",
   "id": "arabic_finance",
   "models_covered": 0,
   "name": "Arabic Finance (HELM Arabic Enterprise)",
   "reasons": [],
   "summary": "HELM's Arabic Enterprise finance set: 299 textbook-derived items in three formats (3-way MCQ, yes/no, numeric calculation), in Arabic or English."
  },
  {
   "aliases": [
    "OALL",
    "Open Arabic LLM Leaderboard",
    "OALL v1"
   ],
   "canonical_id": "arabic_leaderboard_complete",
   "category": "composite",
   "disposition": "unassessed",
   "id": "arabic_leaderboard_complete",
   "models_covered": 0,
   "name": "Open Arabic LLM Leaderboard \u2014 Complete configuration",
   "reasons": [],
   "summary": "An lm-evaluation-harness reproduction of the original Open Arabic LLM Leaderboard: 14 native and machine-translated Arabic task groups aggregated into one size-weighted accuracy score."
  },
  {
   "aliases": [
    "OALL Light"
   ],
   "canonical_id": "arabic_leaderboard_light",
   "category": "composite",
   "disposition": "unassessed",
   "id": "arabic_leaderboard_light",
   "models_covered": 0,
   "name": "Open Arabic LLM Leaderboard \u2014 Light configuration",
   "reasons": [],
   "summary": "A 10%-random-sample version of arabic_leaderboard_complete's same 14 Arabic task groups, run for lower cost; ACVA's 10-item Yemen subset is kept at full size rather than sampled further."
  },
  {
   "aliases": [
    "arabic_legal_qa",
    "arabic_legal_rag",
    "Arabic Enterprise legal"
   ],
   "canonical_id": "arabic_legal",
   "category": "domain",
   "disposition": "unassessed",
   "id": "arabic_legal",
   "models_covered": 0,
   "name": "Arabic Legal (HELM Arabic Enterprise)",
   "reasons": [],
   "summary": "HELM's Arabic Enterprise legal set: 200 UAE-law questions scored by an LLM judge, in closed-book and open-book (statute-in-prompt) modes."
  },
  {
   "aliases": [
    "Arabic MMLU",
    "MBZUAI/ArabicMMLU"
   ],
   "canonical_id": "arabic_mmlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "arabic_mmlu",
   "models_covered": 0,
   "name": "ArabicMMLU",
   "reasons": [],
   "summary": "14,575 native Arabic multiple-choice exam questions across 40 subjects, sourced from school and professional tests in eight countries rather than translated from English MMLU."
  },
  {
   "aliases": [
    "AraDICE"
   ],
   "canonical_id": "aradice",
   "category": "composite",
   "disposition": "unassessed",
   "id": "aradice",
   "models_covered": 0,
   "name": "AraDiCE",
   "reasons": [],
   "summary": "A suite that re-runs six existing English NLU benchmarks in Egyptian, Levantine and Gulf/MSA Arabic, plus a native cultural-knowledge test across six Arab countries."
  },
  {
   "aliases": [
    "Ara Trust",
    "asas-ai/AraTrust"
   ],
   "canonical_id": "aratrust",
   "category": "safety",
   "disposition": "unassessed",
   "id": "aratrust",
   "models_covered": 0,
   "name": "AraTrust",
   "reasons": [],
   "summary": "522 human-written Arabic three-choice trustworthiness questions across eight categories; HELM scores exact match on generated option letters."
  },
  {
   "aliases": [
    "AI2 Reasoning Challenge"
   ],
   "canonical_id": "arc",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "arc",
   "models_covered": 0,
   "name": "ARC (AI2 Reasoning Challenge)",
   "reasons": [],
   "summary": "A grade-school science multiple-choice question set split into Easy and Challenge halves, whose scores are reported interchangeably far too often despite very different difficulty."
  },
  {
   "aliases": [
    "ARC-AGI 2",
    "Abstraction and Reasoning Corpus for AGI, version 2"
   ],
   "canonical_id": "arc_agi_2",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "arc_agi_2",
   "models_covered": 1,
   "name": "ARC-AGI-2",
   "reasons": [],
   "summary": "A grid-puzzle benchmark of novel abstract-reasoning tasks, each solvable by most humans but built to resist memorisation and brute-force search by AI systems."
  },
  {
   "aliases": [
    "AI2 Reasoning Challenge",
    "ARC (Challenge Set)"
   ],
   "canonical_id": "arc_challenge",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "arc_challenge",
   "models_covered": 50,
   "name": "ARC-Challenge",
   "reasons": [],
   "summary": "A multiple-choice grade-school science question set built so that simple retrieval and word-overlap methods fail, isolating genuine multi-hop reasoning."
  },
  {
   "aliases": [
    "ARC (Easy Set)",
    "AI2 Reasoning Challenge (Easy)"
   ],
   "canonical_id": "arc_easy",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "arc_easy",
   "models_covered": 0,
   "name": "ARC-Easy",
   "reasons": [],
   "summary": "The easier, larger half of the AI2 Reasoning Challenge; questions that 2018-era retrieval or word-overlap baselines could already answer, now scored near ceiling by current models."
  },
  {
   "aliases": [
    "ARC-AGI-1 Public Evaluation Set",
    "ARC_Prize_Public_Evaluation",
    "ARC-AGI Public Eval"
   ],
   "canonical_id": "arc_prize_public_evaluation",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "arc_prize_public_evaluation",
   "models_covered": 0,
   "name": "ARC Prize Public Evaluation",
   "reasons": [],
   "summary": "The 400-task public evaluation split of the original ARC-AGI (ARC-AGI-1) grid-puzzle benchmark, fully public since 2019 and now superseded for frontier evaluation by ARC-AGI-2."
  },
  {
   "aliases": [
    "Chatbot Arena",
    "LMArena",
    "Arena"
   ],
   "canonical_id": "arena_elo",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "arena_elo",
   "models_covered": 0,
   "name": "Arena Elo (Chatbot Arena / LMArena)",
   "reasons": [],
   "summary": "Elo-style ratings of chat models derived from live, anonymous, pairwise human-preference votes on real user prompts."
  },
  {
   "aliases": [
    "Chatbot Arena Coding leaderboard",
    "LMArena Coding category"
   ],
   "canonical_id": "arena_elo_coding",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "arena_elo_coding",
   "models_covered": 210,
   "name": "Arena Elo \u2014 Coding",
   "reasons": [],
   "summary": "The Arena text leaderboard's Coding category: Elo ratings computed only from anonymous votes on programming-related prompts."
  },
  {
   "aliases": [
    "Chatbot Arena Hard Prompts",
    "LMArena Hard Prompts category"
   ],
   "canonical_id": "arena_elo_hard_prompts",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "arena_elo_hard_prompts",
   "models_covered": 103,
   "name": "Arena Elo \u2014 Hard Prompts",
   "reasons": [],
   "summary": "An Arena leaderboard built only from votes on prompts an automatic classifier scored as complex and demanding across several hardness criteria."
  },
  {
   "aliases": [
    "Chatbot Arena Math leaderboard",
    "LMArena Math category"
   ],
   "canonical_id": "arena_elo_math",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "arena_elo_math",
   "models_covered": 209,
   "name": "Arena Elo \u2014 Math",
   "reasons": [],
   "summary": "The Arena text leaderboard's Math category: Elo ratings computed only from anonymous votes on maths-related prompts."
  },
  {
   "aliases": [
    "LMArena Overall",
    "Chatbot Arena Overall leaderboard"
   ],
   "canonical_id": "arena_elo_overall",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "arena_elo_overall",
   "models_covered": 211,
   "name": "Arena Elo \u2014 Overall (Text)",
   "reasons": [],
   "summary": "The headline, non-style-controlled Elo ranking of chat models on the Arena (LMArena / Chatbot Arena) text leaderboard."
  },
  {
   "aliases": [
    "Chatbot Arena Style Control",
    "LMArena Style Control",
    "Arena Score, style-controlled"
   ],
   "canonical_id": "arena_elo_style_control",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "arena_elo_style_control",
   "models_covered": 103,
   "name": "Arena Elo \u2014 Style Control",
   "reasons": [],
   "summary": "A style-adjusted Arena ranking that regresses out response length and markdown formatting so ratings lean more on substance than presentation."
  },
  {
   "aliases": [
    "Chatbot Arena Vision leaderboard",
    "Multimodal Arena",
    "LMArena Vision"
   ],
   "canonical_id": "arena_elo_vision",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "arena_elo_vision",
   "models_covered": 55,
   "name": "Arena Elo \u2014 Vision",
   "reasons": [],
   "summary": "A separate Elo leaderboard for vision-language models, built only from Arena battles whose prompt included an image."
  },
  {
   "aliases": [],
   "canonical_id": "arithmetic",
   "category": "math",
   "disposition": "unassessed",
   "id": "arithmetic",
   "models_covered": 0,
   "name": "Arithmetic (GPT-3 synthetic arithmetic tasks)",
   "reasons": [],
   "summary": "Ten fixed synthetic arithmetic tasks -- 2 to 5 digit addition and subtraction, 2-digit multiplication, one-digit composite expressions -- introduced as one small evaluation in the GPT-3 paper."
  },
  {
   "aliases": [
    "AA"
   ],
   "canonical_id": "artificial_analysis",
   "category": "composite",
   "disposition": "unassessed",
   "id": "artificial_analysis",
   "models_covered": 0,
   "name": "Artificial Analysis",
   "reasons": [],
   "summary": "An independent benchmarking company that runs its own model-capability and inference-performance tests and publishes them as composite indices and live leaderboards."
  },
  {
   "aliases": [
    "Artificial Analysis Quality Index",
    "AA Intelligence Index",
    "AAII"
   ],
   "canonical_id": "artificial_analysis_quality_index",
   "category": "composite",
   "disposition": "unassessed",
   "id": "artificial_analysis_quality_index",
   "models_covered": 130,
   "name": "Artificial Analysis Intelligence Index",
   "reasons": [],
   "summary": "Artificial Analysis's own composite capability score, blending ten independently-run evaluations across agents, coding, general knowledge and scientific reasoning into one weighted number."
  },
  {
   "aliases": [
    "Artificial Analysis Speed Index",
    "AA Output Speed",
    "Output Tokens per Second"
   ],
   "canonical_id": "artificial_analysis_speed_index",
   "category": "composite",
   "disposition": "unassessed",
   "id": "artificial_analysis_speed_index",
   "models_covered": 129,
   "name": "Artificial Analysis Output Speed",
   "reasons": [],
   "summary": "Artificial Analysis's live-measured output speed for a model's API: tokens generated per second, a performance measure, not a correctness or quality score."
  },
  {
   "aliases": [],
   "canonical_id": "artificialanalysis_aa_briefcase_lite",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "artificialanalysis_aa_briefcase_lite",
   "models_covered": 0,
   "name": "ArtificialAnalysis/AA-Briefcase-Lite",
   "reasons": [],
   "summary": "ArtificialAnalysis/AA-Briefcase-Lite is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "artificialanalysis_aa_lcr",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "artificialanalysis_aa_lcr",
   "models_covered": 0,
   "name": "ArtificialAnalysis/AA-LCR",
   "reasons": [],
   "summary": "ArtificialAnalysis/AA-LCR is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [
    "AA-Omniscience Public",
    "AA-Omniscience Accuracy"
   ],
   "canonical_id": "artificialanalysis_aa_omniscience_public",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "artificialanalysis_aa_omniscience_public",
   "models_covered": 0,
   "name": "AA-Omniscience",
   "reasons": [],
   "summary": "AA-Omniscience evaluates factual knowledge and hallucination behavior across public-domain questions in several professional and academic domains."
  },
  {
   "aliases": [
    "ArxivRoll",
    "RoBench"
   ],
   "canonical_id": "arxivrollbench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "arxivrollbench",
   "models_covered": 0,
   "name": "ArxivRollBench",
   "reasons": [],
   "summary": "A rolling benchmark turning freshly-published arXiv text into sentence-ordering, cloze and next-fragment multiple-choice tasks every six months, built to measure how much contamination inflates benchmark scores."
  },
  {
   "aliases": [
    "ascii-word-recognition"
   ],
   "canonical_id": "ascii_word_recognition",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "ascii_word_recognition",
   "models_covered": 0,
   "name": "ASCII Word Recognition (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench task: name the English word shown as ASCII art in one of five Figlet fonts."
  },
  {
   "aliases": [
    "ASDiv",
    "ASDIV",
    "Academia Sinica Diverse MWP Dataset",
    "nlu-asdiv-dataset"
   ],
   "canonical_id": "asdiv",
   "category": "math",
   "disposition": "unassessed",
   "id": "asdiv",
   "models_covered": 0,
   "name": "ASDiv",
   "reasons": [],
   "summary": "2,305 elementary English math word problems with annotated type and grade; lm-eval runs a log-likelihood task and an 8-shot GSM8K-style CoT variant."
  },
  {
   "aliases": [],
   "canonical_id": "assistant_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "assistant_bench",
   "models_covered": 0,
   "name": "AssistantBench",
   "reasons": [],
   "summary": "214 realistic web tasks such as monitoring listings or comparing prices, scored by an automatic answer-matching function; no system has cleared half credit on the public leaderboard."
  },
  {
   "aliases": [
    "Assistant Skills in Tool-use, Reasoning & Action-planning",
    "ASTRA-bench (Apple)"
   ],
   "canonical_id": "astra_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "astra_bench",
   "models_covered": 0,
   "name": "ASTRA-bench",
   "reasons": [],
   "summary": "Scores personal-assistant tool use on 2,413 scenarios that mix a stateful email-calendar-messaging sandbox with time-evolving synthetic user context."
  },
  {
   "aliases": [
    "AbSTRAct Question Answering over documents"
   ],
   "canonical_id": "astra_qa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "astra_qa",
   "models_covered": 0,
   "name": "ASTRA-QA",
   "reasons": [],
   "summary": "Scores whether a RAG system covers required topics and avoids curated unsupported claims on 869 abstract questions over papers and news."
  },
  {
   "aliases": [
    "ATLAS",
    "AGI-Oriented Testbed for Logical Application in Science"
   ],
   "canonical_id": "atlas",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "atlas",
   "models_covered": 0,
   "name": "ATLAS (AGI-Oriented Testbed for Logical Application in Science)",
   "reasons": [],
   "summary": "798 expert-written STEM problems across seven fields, scored by an LLM judge; GPT-5-High is at 42.9% on the public validation set."
  },
  {
   "aliases": [
    "Authorship Verification in a Swapping Scenario"
   ],
   "canonical_id": "authorship_verification",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "authorship_verification",
   "models_covered": 0,
   "name": "Authorship Verification (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench two-choice task: match a ~500-word Gutenberg passage to its author, in same-genre and genre-swapped pairings."
  },
  {
   "aliases": [
    "Autoclassification of Words/Objects"
   ],
   "canonical_id": "auto_categorization",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "auto_categorization",
   "models_covered": 0,
   "name": "Auto Categorization",
   "reasons": [],
   "summary": "A 328-item BIG-bench free-response task that names the shared category of a short list of words or objects."
  },
  {
   "aliases": [
    "Program State Analysis/Automatic Debugging",
    "Program State Analysis"
   ],
   "canonical_id": "auto_debugging",
   "category": "coding",
   "disposition": "unassessed",
   "id": "auto_debugging",
   "models_covered": 0,
   "name": "Auto Debugging",
   "reasons": [],
   "summary": "A 34-item BIG-bench Lite task that asks for a Python program's intermediate state or exception without running the code."
  },
  {
   "aliases": [],
   "canonical_id": "autobencher_capabilities",
   "category": "composite",
   "disposition": "unassessed",
   "id": "autobencher_capabilities",
   "models_covered": 0,
   "name": "AutoBencher Capabilities",
   "reasons": [],
   "summary": "AutoBencher Capabilities is a 2,377-question HELM benchmark whose math, history, science, economics and multilingual QA items were searched for and generated by a language model, not written by people."
  },
  {
   "aliases": [],
   "canonical_id": "autobencher_safety",
   "category": "safety",
   "disposition": "unassessed",
   "id": "autobencher_safety",
   "models_covered": 0,
   "name": "AutoBencher Safety",
   "reasons": [],
   "summary": "AutoBencher Safety is a 301-prompt HELM refusal test whose harmful requests across 30 topics were searched for and generated by a language model to maximize how often models comply, not written by people."
  },
  {
   "aliases": [
    "Zapier AutomationBench",
    "zapier/AutomationBench"
   ],
   "canonical_id": "automationbench",
   "category": "agentic",
   "disposition": "unverified",
   "id": "automationbench",
   "models_covered": 0,
   "name": "AutomationBench",
   "reasons": [
    "no qualifying current frontier/open coverage from different organizations",
    "review is not approved",
    "task, metric, and protocol are incomplete",
    "usefulness is not established",
    "usefulness is unknown"
   ],
   "summary": "Zapier's agentic benchmark of cross-app business workflows, scored by whether simulated SaaS state matches every assertion after REST-API tool use."
  },
  {
   "aliases": [
    "AA AutomationBench",
    "AutomationBench AA"
   ],
   "canonical_id": "automationbench_aa",
   "category": "agentic",
   "disposition": "active",
   "id": "automationbench_aa",
   "models_covered": 0,
   "name": "AutomationBench-AA",
   "reasons": [
    "qualifying current coverage"
   ],
   "summary": "Artificial Analysis's independent run of Zapier's private AutomationBench split, scoring objective completion with zero credit after any guardrail break."
  },
  {
   "aliases": [
    "bAbI",
    "bAbI tasks",
    "Facebook bAbI"
   ],
   "canonical_id": "babi_qa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "babi_qa",
   "models_covered": 0,
   "name": "bAbI (Question-Answering Tasks)",
   "reasons": [],
   "summary": "20 synthetic reading-comprehension toy tasks, from single-fact retrieval to induction and path-finding, that check basic reasoning skills; long saturated, now mainly the substrate for BABILong."
  },
  {
   "aliases": [
    "bAbILong"
   ],
   "canonical_id": "babilong",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "babilong",
   "models_covered": 0,
   "name": "BABILong",
   "reasons": [],
   "summary": "Embeds the 20 classic bAbI reasoning tasks as needles inside long PG19 book text, testing whether models can find and combine scattered facts at context lengths up to millions of tokens."
  },
  {
   "aliases": [
    "bangla_boolQA",
    "BoolQ Bangla",
    "BoolQ-BN"
   ],
   "canonical_id": "bangla_boolqa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "bangla_boolqa",
   "models_covered": 0,
   "name": "Bangla BoolQA",
   "reasons": [],
   "summary": "1,976 Bangla yes/no reading-comprehension questions GPT-4-generated from Bangla Wikipedia, Banglapedia and news passages; inspired by BoolQ's format but not a translation of it."
  },
  {
   "aliases": [
    "bangla_commonsenseQA",
    "CommonsenseQA-BN",
    "CommonsenseQA Bangla"
   ],
   "canonical_id": "bangla_commonsenseqa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "bangla_commonsenseqa",
   "models_covered": 0,
   "name": "Bangla CommonsenseQA",
   "reasons": [],
   "summary": "10,962-item machine translation of CommonsenseQA into Bangla, built with an automated Google-Translate-plus-LLM-rewriting pipeline the authors call Expressive Semantic Translation."
  },
  {
   "aliases": [
    "BN MMLU",
    "titulm-bangla-mmlu",
    "TituLLM Bangla MMLU"
   ],
   "canonical_id": "bangla_mmlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "bangla_mmlu",
   "models_covered": 0,
   "name": "Bangla MMLU",
   "reasons": [],
   "summary": "87,869 Bangla four-choice exam questions from Bangladeshi admissions, HSC and job tests; lm-eval scores the 14,750-item test split."
  },
  {
   "aliases": [
    "bangla_poenbookQA",
    "bangla_openbookQA",
    "OpenBookQA-BN",
    "OpenBookQA Bangla"
   ],
   "canonical_id": "bangla_openbookqa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "bangla_openbookqa",
   "models_covered": 0,
   "name": "Bangla OpenBookQA",
   "reasons": [],
   "summary": "5,944-item machine translation of OpenBookQA into Bangla, built with an automated Google-Translate-plus-LLM-rewriting pipeline the authors call Expressive Semantic Translation."
  },
  {
   "aliases": [
    "bangla_piQA",
    "PIQA-BN",
    "PIQA Bangla"
   ],
   "canonical_id": "bangla_piqa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "bangla_piqa",
   "models_covered": 0,
   "name": "Bangla PIQA",
   "reasons": [],
   "summary": "17,177-item machine translation of PIQA's physical-commonsense-reasoning questions into Bangla, built with an automated Google-Translate-plus-LLM-rewriting pipeline the authors call EST."
  },
  {
   "aliases": [
    "BANKING 77",
    "PolyAI BANKING77"
   ],
   "canonical_id": "banking77",
   "category": "domain",
   "disposition": "unassessed",
   "id": "banking77",
   "models_covered": 0,
   "name": "BANKING77",
   "reasons": [],
   "summary": "English banking intent classification of 13,083 customer-service queries into 77 intents; HELM scores generated labels by exact match."
  },
  {
   "aliases": [],
   "canonical_id": "basque_bench",
   "category": "composite",
   "disposition": "unassessed",
   "id": "basque_bench",
   "models_covered": 0,
   "name": "BasqueBench",
   "reasons": [],
   "summary": "18 Basque-language tasks -- reused Basque NLU/QA sets plus six built for this suite -- bundled into one lm-evaluation-harness group to score base LLMs on Basque, part of the wider IberoBench project."
  },
  {
   "aliases": [
    "Basque GLUE",
    "orai-nlp/basqueGLUE"
   ],
   "canonical_id": "basqueglue",
   "category": "composite",
   "disposition": "unassessed",
   "id": "basqueglue",
   "models_covered": 0,
   "name": "BasqueGLUE",
   "reasons": [],
   "summary": "Nine-task Basque NLU suite in the GLUE mould, spanning NER, dialogue, topic, sentiment, stance, QNLI, word-in-context and coreference."
  },
  {
   "aliases": [
    "BBEH"
   ],
   "canonical_id": "bbeh",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "bbeh",
   "models_covered": 0,
   "name": "BIG-Bench Extra Hard (BBEH)",
   "reasons": [],
   "summary": "Replaces each of BBH's 23 tasks with a substantially harder variant of the same reasoning skill, calibrated so two strong 2025 reference models both scored under 70%."
  },
  {
   "aliases": [
    "BBH"
   ],
   "canonical_id": "bbh",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "bbh",
   "models_covered": 221,
   "name": "BIG-Bench Hard",
   "reasons": [],
   "summary": "A 23-task suite pulled from BIG-Bench specifically because prior language models failed to beat average human raters on them."
  },
  {
   "aliases": [
    "Bias Benchmark for Question Answering"
   ],
   "canonical_id": "bbq",
   "category": "safety",
   "disposition": "unassessed",
   "id": "bbq",
   "models_covered": 66,
   "name": "BBQ (Bias Benchmark for QA)",
   "reasons": [],
   "summary": "Multiple-choice QA benchmark testing whether models default to social stereotypes under ambiguous context and can override them once context disambiguates the answer."
  },
  {
   "aliases": [
    "BBQ Lite",
    "Bias Benchmark for QA Lite"
   ],
   "canonical_id": "bbq_lite",
   "category": "safety",
   "disposition": "unassessed",
   "id": "bbq_lite",
   "models_covered": 0,
   "name": "BBQ-Lite (Bias Benchmark for QA, BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench BBQ subset: 16,076 three-choice US-English QA items that test social stereotypes in ambiguous and disambiguated contexts."
  },
  {
   "aliases": [
    "BBQ Lite JSON",
    "bbq-lite-json"
   ],
   "canonical_id": "bbq_lite_json",
   "category": "safety",
   "disposition": "unassessed",
   "id": "bbq_lite_json",
   "models_covered": 0,
   "name": "BBQ-Lite JSON (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench JSON task with 16,076 three-way QA items on US social stereotypes; scores only multiple-choice grade, not BBQ's bias score."
  },
  {
   "aliases": [
    "BEIR: A Heterogenous Benchmark for Zero-shot Evaluation of Information Retrieval Models"
   ],
   "canonical_id": "beir",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "beir",
   "models_covered": 69,
   "name": "BEIR",
   "reasons": [],
   "summary": "A suite of 18 public retrieval datasets across 9 task types used to test whether a search or embedding model generalises to new domains without fine-tuning."
  },
  {
   "aliases": [
    "The Belebele Benchmark"
   ],
   "canonical_id": "belebele",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "belebele",
   "models_covered": 0,
   "name": "Belebele",
   "reasons": [],
   "summary": "Parallel four-way reading-comprehension set: 900 questions in each of 122 language variants, 109,800 items, passages from FLORES-200."
  },
  {
   "aliases": [
    "tau-bench"
   ],
   "canonical_id": "bench_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "bench_bench",
   "models_covered": 0,
   "name": "\u03c4-bench",
   "reasons": [],
   "summary": "\u03c4-bench evaluates whether tool-using agents complete user goals while following domain policies across stateful conversations."
  },
  {
   "aliases": [],
   "canonical_id": "bench_coe",
   "category": "composite",
   "disposition": "unassessed",
   "id": "bench_coe",
   "models_covered": 0,
   "name": "Bench-CoE",
   "reasons": [],
   "summary": "Bench-CoE evaluates routing and collaboration among specialist language and multimodal experts using benchmark-derived training data."
  },
  {
   "aliases": [],
   "canonical_id": "bench_mfg",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "bench_mfg",
   "models_covered": 0,
   "name": "Bench-MFG",
   "reasons": [],
   "summary": "Bench-MFG evaluates algorithms for learning stationary mean-field games across standardized discrete environments and generated instances."
  },
  {
   "aliases": [],
   "canonical_id": "benchmark_bcplus",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "benchmark_bcplus",
   "models_covered": 0,
   "name": "benchmark-bcplus",
   "reasons": [],
   "summary": "benchmark-bcplus is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "benchmark_contamination",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "benchmark_contamination",
   "models_covered": 0,
   "name": "Benchmark Contamination",
   "reasons": [],
   "summary": "Benchmark Contamination is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "benchmark_research",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "benchmark_research",
   "models_covered": 0,
   "name": "benchmark-research",
   "reasons": [],
   "summary": "benchmark-research is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "benchmark_results",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "benchmark_results",
   "models_covered": 0,
   "name": "benchmark_results",
   "reasons": [],
   "summary": "benchmark_results is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [
    "MINT",
    "Medical Incremental N-Turn Benchmark",
    "Benchmarking Multi-turn Medical Diagnosis"
   ],
   "canonical_id": "benchmarking_multi_turn_medical_diagnosis",
   "category": "domain",
   "disposition": "unassessed",
   "id": "benchmarking_multi_turn_medical_diagnosis",
   "models_covered": 0,
   "name": "MINT (Medical Incremental N-Turn Benchmark)",
   "reasons": [],
   "summary": "MINT shards 1,035 medical cases into multi-turn evidence to test whether models diagnose too early, self-correct, or get lured by lab results."
  },
  {
   "aliases": [],
   "canonical_id": "benchmarking_the_benchmarks",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "benchmarking_the_benchmarks",
   "models_covered": 0,
   "name": "Benchmarking the Benchmarks",
   "reasons": [],
   "summary": "Benchmarking the Benchmarks tests whether commonsense benchmark rankings predict performance on downstream social, pragmatic, temporal, and physical reasoning tasks."
  },
  {
   "aliases": [],
   "canonical_id": "benchmarking_the_domain_gap",
   "category": "domain",
   "disposition": "unassessed",
   "id": "benchmarking_the_domain_gap",
   "models_covered": 0,
   "name": "Benchmarking the Domain Gap",
   "reasons": [],
   "summary": "Benchmarking the Domain Gap measures how model rankings change across video capsule endoscopy datasets and shared-label targets."
  },
  {
   "aliases": [],
   "canonical_id": "benchmarking_the_residual",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "benchmarking_the_residual",
   "models_covered": 0,
   "name": "Benchmarking the Residual",
   "reasons": [],
   "summary": "Benchmarking the Residual proposes a horizon residual for separating ordinary stage errors from degradation caused by long-horizon execution."
  },
  {
   "aliases": [],
   "canonical_id": "bertaqa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "bertaqa",
   "models_covered": 0,
   "name": "BertaQA",
   "reasons": [],
   "summary": "4,756 three-choice trivia questions, parallel in Basque and English and split almost evenly between Basque-local and general-global topics, to isolate what a model knows about one specific culture."
  },
  {
   "aliases": [],
   "canonical_id": "beyond_bleu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "beyond_bleu",
   "models_covered": 0,
   "name": "Beyond BLEU",
   "reasons": [],
   "summary": "Beyond BLEU is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "beyond_flops",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "beyond_flops",
   "models_covered": 0,
   "name": "Beyond FLOPs",
   "reasons": [],
   "summary": "Beyond FLOPs is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "beyond_leaderboards",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "beyond_leaderboards",
   "models_covered": 0,
   "name": "Beyond Leaderboards",
   "reasons": [],
   "summary": "Beyond Leaderboards is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "beyond_mse",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "beyond_mse",
   "models_covered": 0,
   "name": "Beyond MSE",
   "reasons": [],
   "summary": "Beyond MSE is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "beyondaime",
   "category": "math",
   "disposition": "unassessed",
   "id": "beyondaime",
   "models_covered": 0,
   "name": "BeyondAIME",
   "reasons": [],
   "summary": "100 newly written competition math problems at or above the difficulty of AIME's hardest five problems, each manually revised to be unique and to resist guessing, with a single verifiable integer answer."
  },
  {
   "aliases": [
    "Berkeley Function-Calling Leaderboard",
    "Berkeley Function Calling Leaderboard",
    "BFCL"
   ],
   "canonical_id": "bfcl",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "bfcl",
   "models_covered": 0,
   "name": "BFCL (Berkeley Function-Calling Leaderboard)",
   "reasons": [],
   "summary": "UC Berkeley leaderboard for LLM function calling: single-turn AST match, live user data, multi-turn backends, and agentic memory/web-search tasks."
  },
  {
   "aliases": [
    "BHS",
    "Controlled Evaluation of Syntactic Knowledge"
   ],
   "canonical_id": "bhs",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "bhs",
   "models_covered": 0,
   "name": "BHS (Basque, Hindi, Swahili syntactic evaluation)",
   "reasons": [],
   "summary": "22 suites of 1,000 minimal pairs each that test whether models prefer grammatical Basque, Hindi, or Swahili continuations."
  },
  {
   "aliases": [
    "bias from probabilities",
    "Social Bias from Sentence Probability"
   ],
   "canonical_id": "bias_from_probabilities",
   "category": "safety",
   "disposition": "unassessed",
   "id": "bias_from_probabilities",
   "models_covered": 0,
   "name": "Social Bias from Sentence Probability (BIG-bench)",
   "reasons": [],
   "summary": "Programmatic BIG-bench task that scores worst-case shifts in sentence probability when race, religion, or gender words are swapped in templates."
  },
  {
   "aliases": [
    "BIG-Bench",
    "Beyond the Imitation Game Benchmark"
   ],
   "canonical_id": "big_bench",
   "category": "composite",
   "disposition": "unassessed",
   "id": "big_bench",
   "models_covered": 0,
   "name": "BIG-bench (Beyond the Imitation Game Benchmark)",
   "reasons": [],
   "summary": "A collaborative suite of 200-plus wildly varied tasks from 450 authors; almost every number called BIG-bench today is really BIG-bench Hard or BIG-bench Lite."
  },
  {
   "aliases": [],
   "canonical_id": "bigcodebench",
   "category": "coding",
   "disposition": "unassessed",
   "id": "bigcodebench",
   "models_covered": 0,
   "name": "BigCodeBench",
   "reasons": [],
   "summary": "1,140 Python tasks that require chaining calls across 139 real libraries, testing whether a model can use diverse tools correctly rather than write self-contained algorithmic code."
  },
  {
   "aliases": [
    "biology-instruction"
   ],
   "canonical_id": "biodata",
   "category": "domain",
   "disposition": "unassessed",
   "id": "biodata",
   "models_covered": 0,
   "name": "OpenCompass biodata (biology-instruction)",
   "reasons": [],
   "summary": "OpenCompass suite of 21 biology tasks on opencompass/biology-instruction, scored with MCC, correlation, R\u00b2, AUC, accuracy and EC-number Fmax."
  },
  {
   "aliases": [],
   "canonical_id": "bird",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "bird",
   "models_covered": 0,
   "name": "BIRD",
   "reasons": [],
   "summary": "BIRD is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "bird_critic",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "bird_critic",
   "models_covered": 0,
   "name": "BIRD-CRITIC",
   "reasons": [],
   "summary": "BIRD-CRITIC is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "bird_history",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "bird_history",
   "models_covered": 0,
   "name": "BIRD-History",
   "reasons": [],
   "summary": "BIRD-History is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "bird_interact",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "bird_interact",
   "models_covered": 0,
   "name": "BIRD-INTERACT",
   "reasons": [],
   "summary": "BIRD-INTERACT is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [
    "BIRD",
    "BIRD-SQL",
    "BIRD SQL"
   ],
   "canonical_id": "bird_sql",
   "category": "coding",
   "disposition": "unassessed",
   "id": "bird_sql",
   "models_covered": 0,
   "name": "BIRD-SQL (HELM bird_sql / BIRD Dev)",
   "reasons": [],
   "summary": "Text-to-SQL over 95 large databases; HELM's bird_sql run is the public 1,534-item development split scored by execution accuracy."
  },
  {
   "aliases": [
    "Benchmark of Linguistic Minimal Pairs for English"
   ],
   "canonical_id": "blimp",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "blimp",
   "models_covered": 0,
   "name": "BLiMP (Benchmark of Linguistic Minimal Pairs)",
   "reasons": [],
   "summary": "67 automatically generated 1,000-pair paradigms testing whether a model's probabilities favour the grammatical member of a minimal sentence pair -- linguistic knowledge, not task-solving ability."
  },
  {
   "aliases": [
    "BLiMP-NL",
    "BLiMP-NL large"
   ],
   "canonical_id": "blimp_nl",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "blimp_nl",
   "models_covered": 0,
   "name": "BLiMP-NL (Benchmark of Linguistic Minimal Pairs for Dutch)",
   "reasons": [],
   "summary": "8,400 Dutch minimal pairs across 84 paradigms and 22 phenomena, scored by whether a model prefers the grammatical sentence over a close ungrammatical match."
  },
  {
   "aliases": [
    "Brazilian Leading Universities Entrance eXams"
   ],
   "canonical_id": "bluex",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "bluex",
   "models_covered": 0,
   "name": "BLUEX (Brazilian Leading Universities Entrance eXams)",
   "reasons": [],
   "summary": "Portuguese multiple-choice questions from Unicamp (Convest) and USP (Fuvest) entrance exams, including image-linked items that most text-only harnesses drop."
  },
  {
   "aliases": [
    "Bias in Open-Ended Language Generation Dataset"
   ],
   "canonical_id": "bold",
   "category": "safety",
   "disposition": "unassessed",
   "id": "bold",
   "models_covered": 0,
   "name": "BOLD (Bias in Open-Ended Language Generation Dataset)",
   "reasons": [],
   "summary": "23,679 Wikipedia-derived prompts across five demographic domains, used to check whether a model's open-ended completions differ in sentiment, toxicity or regard depending on the group named in the prompt."
  },
  {
   "aliases": [
    "BBH boolean_expressions"
   ],
   "canonical_id": "boolean_expressions",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "boolean_expressions",
   "models_covered": 0,
   "name": "Boolean Expressions (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench task that generates random True/False expressions with and, or, and not, then scores whether the model prefers the correct truth value."
  },
  {
   "aliases": [
    "Boolean Questions"
   ],
   "canonical_id": "boolq",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "boolq",
   "models_covered": 0,
   "name": "BoolQ",
   "reasons": [],
   "summary": "15,942 naturally occurring yes/no reading-comprehension questions paired with a Wikipedia passage; part of SuperGLUE and now largely saturated for frontier models."
  },
  {
   "aliases": [
    "BARQA-ISNotes",
    "BARQA"
   ],
   "canonical_id": "bridging_anaphora_resolution_barqa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "bridging_anaphora_resolution_barqa",
   "models_covered": 0,
   "name": "Bridging Anaphora Resolution as Question Answering (BARQA-ISNotes)",
   "reasons": [],
   "summary": "648 questions built from 50 Wall Street Journal articles that ask a model to name the implicit antecedent behind a bridging phrase, such as \"limited access of what?\", using only the preceding context."
  },
  {
   "aliases": [
    "BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents"
   ],
   "canonical_id": "browsecomp",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "browsecomp",
   "models_covered": 3,
   "name": "BrowseComp",
   "reasons": [],
   "summary": "1,266 deliberately hard-to-find, easy-to-verify questions that measure whether an agent can persistently search the web to pin down a single fact."
  },
  {
   "aliases": [],
   "canonical_id": "buysidefinbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "buysidefinbench",
   "models_covered": 0,
   "name": "BuySideFinBench",
   "reasons": [],
   "summary": "A single-contributor, bilingual Chinese/English OpenCompass benchmark of 180 multiple-choice questions testing buy-side equity-research skills such as DCF valuation and three-statement linkage."
  },
  {
   "aliases": [
    "Catalan Bias Benchmark for Question Answering",
    "Catalan BBQ"
   ],
   "canonical_id": "cabbq",
   "category": "safety",
   "disposition": "unassessed",
   "id": "cabbq",
   "models_covered": 0,
   "name": "CaBBQ (Catalan Bias Benchmark for Question Answering)",
   "reasons": [],
   "summary": "Catalan multiple-choice QA, parallel with EsBBQ, testing stereotype use under ambiguous context and accuracy once the context names the answer."
  },
  {
   "aliases": [
    "CaLM Lite",
    "Causal Evaluation of Language Models"
   ],
   "canonical_id": "calm",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "calm",
   "models_covered": 0,
   "name": "CaLM (Causal Evaluation of Language Models)",
   "reasons": [],
   "summary": "OpenCompass `calm` runs CaLM Lite: 9,200 English and Chinese items over 92 causal targets, a tenth of the full 126,334-sample CaLM suite."
  },
  {
   "aliases": [],
   "canonical_id": "cardbiomedbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "cardbiomedbench",
   "models_covered": 0,
   "name": "CARDBiomedBench",
   "reasons": [],
   "summary": "An NIH biomedical-research QA benchmark of 68,227 expert- and template-generated questions on neurodegenerative-disease genetics, molecular biology and clinical knowledge, LLM-judged for quality and safety."
  },
  {
   "aliases": [],
   "canonical_id": "careqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "careqa",
   "models_covered": 0,
   "name": "CareQA",
   "reasons": [],
   "summary": "5,621 closed multiple-choice healthcare questions from Spain's 2020-2024 specialised exams, in English and Spanish, plus a 2,769-item open-ended English variant scored by a new metric."
  },
  {
   "aliases": [
    "Case Holdings On Legal Decisions"
   ],
   "canonical_id": "casehold",
   "category": "domain",
   "disposition": "unassessed",
   "id": "casehold",
   "models_covered": 0,
   "name": "CaseHOLD (Case Holdings On Legal Decisions)",
   "reasons": [],
   "summary": "53,137 five-way multiple-choice questions that ask which holding statement matches a citing passage mined from US judicial opinions."
  },
  {
   "aliases": [],
   "canonical_id": "catalan_bench",
   "category": "composite",
   "disposition": "unassessed",
   "id": "catalan_bench",
   "models_covered": 0,
   "name": "CatalanBench",
   "reasons": [],
   "summary": "An EleutherAI lm-evaluation-harness suite aggregating roughly two dozen Catalan- and Valencian-language tasks -- QA, NLI, commonsense, paraphrase, summarisation and translation -- built and funded by BSC's Projecte AINA."
  },
  {
   "aliases": [
    "causal_judgement",
    "BBH causal_judgement"
   ],
   "canonical_id": "causal_judgment",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "causal_judgment",
   "models_covered": 0,
   "name": "Causal Judgment (BIG-bench)",
   "reasons": [],
   "summary": "190 yes/no questions, from psychology papers, asking whether a typical person would say X caused Y in a short moral or counterfactual story."
  },
  {
   "aliases": [
    "BIG-bench cause_and_effect"
   ],
   "canonical_id": "cause_and_effect",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "cause_and_effect",
   "models_covered": 0,
   "name": "Cause and Effect",
   "reasons": [],
   "summary": "A 51-pair BIG-bench task that asks which of two English events caused the other, scored as two-way multiple choice under three prompt formats."
  },
  {
   "aliases": [],
   "canonical_id": "ceval",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "ceval",
   "models_covered": 0,
   "name": "C-Eval",
   "reasons": [],
   "summary": "A 13,948-question, four-option Chinese exam benchmark across 52 subjects and four difficulty levels, whose test set was held out via a submission site until a full public release in July 2025."
  },
  {
   "aliases": [
    "CHARM"
   ],
   "canonical_id": "charm",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "charm",
   "models_covered": 0,
   "name": "CHARM (Benchmarking Chinese Commonsense Reasoning of LLMs)",
   "reasons": [],
   "summary": "Tests Chinese-specific commonsense reasoning against a matched globally-known-commonsense control, plus free-form memorization questions built from the same underlying facts."
  },
  {
   "aliases": [
    "ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning"
   ],
   "canonical_id": "chartqa",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "chartqa",
   "models_covered": 64,
   "name": "ChartQA",
   "reasons": [],
   "summary": "Visual question answering over real-world chart images that require reading values off the chart and doing arithmetic or logical reasoning to answer."
  },
  {
   "aliases": [
    "CharXiv: Charting Gaps in Realistic Chart Understanding in Multimodal LLMs"
   ],
   "canonical_id": "charxiv",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "charxiv",
   "models_covered": 0,
   "name": "CharXiv",
   "reasons": [],
   "summary": "2,323 hand-picked charts from arXiv papers, paired with descriptive and reasoning questions built to resist the score inflation seen on template-based chart benchmarks."
  },
  {
   "aliases": [
    "CharXiv-R"
   ],
   "canonical_id": "charxiv_reasoning",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "charxiv_reasoning",
   "models_covered": 3,
   "name": "CharXiv Reasoning",
   "reasons": [],
   "summary": "The reasoning half of CharXiv: one open-ended question per chart that requires combining several visual elements, not just reading a labelled value."
  },
  {
   "aliases": [
    "CharXiv-R w/ tools"
   ],
   "canonical_id": "charxiv_reasoning_tools",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "charxiv_reasoning_tools",
   "models_covered": 2,
   "name": "CharXiv Reasoning (with tool use)",
   "reasons": [],
   "summary": "CharXiv Reasoning scored with the model given a tool during the eval -- an image-cropping tool in the one vendor report this page found -- instead of the static chart image alone."
  },
  {
   "aliases": [
    "Checkmate In One Move",
    "BIG-bench checkmate_in_one"
   ],
   "canonical_id": "checkmate_in_one",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "checkmate_in_one",
   "models_covered": 0,
   "name": "Checkmate in One",
   "reasons": [],
   "summary": "A 3,500-item BIG-bench task that asks for the unique checkmate-in-one move in standard algebraic notation after a Lichess game prefix."
  },
  {
   "aliases": [
    "Chem_exam"
   ],
   "canonical_id": "chem_exam",
   "category": "domain",
   "disposition": "unassessed",
   "id": "chem_exam",
   "models_covered": 0,
   "name": "Chem Exam",
   "reasons": [],
   "summary": "An OpenCompass-only chemistry benchmark of gaokao-style exam and competition problems, LLM-judge scored for partial credit; no paper, author or publicly locatable dataset was found."
  },
  {
   "aliases": [],
   "canonical_id": "chembench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "chembench",
   "models_covered": 0,
   "name": "ChemBench",
   "reasons": [],
   "summary": "An automated chemistry benchmark of nearly 2,800 question-answer pairs, built specifically to compare frontier LLMs against surveyed human chemists rather than just each other."
  },
  {
   "aliases": [
    "State Tracking in Chess",
    "BIG-bench chess_state_tracking"
   ],
   "canonical_id": "chess_state_tracking",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "chess_state_tracking",
   "models_covered": 0,
   "name": "Chess State Tracking",
   "reasons": [],
   "summary": "A 6,000-item BIG-bench task that, given a UCI game prefix and a start square, asks for any legal destination square of that piece."
  },
  {
   "aliases": [
    "CRT",
    "BIG-bench chinese_remainder_theorem"
   ],
   "canonical_id": "chinese_remainder_theorem",
   "category": "math",
   "disposition": "unassessed",
   "id": "chinese_remainder_theorem",
   "models_covered": 0,
   "name": "Chinese Remainder Theorem",
   "reasons": [],
   "summary": "A 500-item BIG-bench task that asks for the unique integer satisfying three coprime remainder conditions, paraphrased as an English word problem."
  },
  {
   "aliases": [
    "Chinese-SimpleQA"
   ],
   "canonical_id": "chinese_simpleqa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "chinese_simpleqa",
   "models_covered": 0,
   "name": "Chinese SimpleQA",
   "reasons": [],
   "summary": "The Chinese counterpart to OpenAI's SimpleQA -- 3,000 short fact-seeking questions across 6 topics and 99 subtopics, from Alibaba's Taobao & Tmall Group, LLM-graded following SimpleQA's approach."
  },
  {
   "aliases": [
    "NoteExtract",
    "CHW Care Plan",
    "chw_care_plan"
   ],
   "canonical_id": "chw_care_plan",
   "category": "domain",
   "disposition": "unassessed",
   "id": "chw_care_plan",
   "models_covered": 0,
   "name": "NoteExtract (MedHELM chw_care_plan)",
   "reasons": [],
   "summary": "MedHELM private NoteExtract task: rewrite a community health worker care-plan note into a fixed clinical template, scored by an LLM jury."
  },
  {
   "aliases": [],
   "canonical_id": "ci_mcqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "ci_mcqa",
   "models_covered": 0,
   "name": "CIMCQA",
   "reasons": [],
   "summary": "A HELM scenario for CS-education concept-inventory multiple-choice questions that cannot be run outside its authors because the item data is private."
  },
  {
   "aliases": [],
   "canonical_id": "cibench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "cibench",
   "models_covered": 0,
   "name": "CIBench",
   "reasons": [],
   "summary": "OpenCompass's interactive benchmark for LLM code-interpreter agents -- 234 multi-step data-science tasks (1,900+ questions) across ten Python libraries, scored end-to-end and in an error-corrected oracle mode."
  },
  {
   "aliases": [
    "CIFAR-10 Test",
    "BIG-bench cifar10_classification"
   ],
   "canonical_id": "cifar10_classification",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "cifar10_classification",
   "models_covered": 0,
   "name": "CIFAR-10 Classification (BIG-bench encodings)",
   "reasons": [],
   "summary": "A 20,000-item BIG-bench task that classifies CIFAR-10 test images from base64 PNG strings or hex pixel arrays, without a vision encoder."
  },
  {
   "aliases": [
    "CivilComments",
    "CivilComments-wilds",
    "Jigsaw Civil Comments"
   ],
   "canonical_id": "civil_comments",
   "category": "safety",
   "disposition": "unassessed",
   "id": "civil_comments",
   "models_covered": 0,
   "name": "CivilComments (HELM)",
   "reasons": [],
   "summary": "HELM wrap of WILDS CivilComments: the model reads an English comment and answers True or False to whether it is toxic, also sliced by identity group."
  },
  {
   "aliases": [],
   "canonical_id": "cl_bench_a_benchmark_for_context_learning",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "cl_bench_a_benchmark_for_context_learning",
   "models_covered": 0,
   "name": "CL-bench",
   "reasons": [],
   "summary": "CL-bench tests whether models can learn new domain knowledge, rules and procedures from complex contexts."
  },
  {
   "aliases": [],
   "canonical_id": "cl_bench_life",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "cl_bench_life",
   "models_covered": 0,
   "name": "CL-bench Life",
   "reasons": [],
   "summary": "CL-bench Life evaluates context learning over messy, fragmented real-life contexts such as conversations, archives and behavioral traces."
  },
  {
   "aliases": [],
   "canonical_id": "class_eval",
   "category": "coding",
   "disposition": "unassessed",
   "id": "class_eval",
   "models_covered": 0,
   "name": "ClassEval",
   "reasons": [],
   "summary": "100 hand-written Python class-generation tasks testing whether a model can implement a whole class correctly, not just a single function like HumanEval."
  },
  {
   "aliases": [
    "CLBench",
    "CL-bench: A Benchmark for Context Learning"
   ],
   "canonical_id": "clbench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "clbench",
   "models_covered": 0,
   "name": "CL-bench",
   "reasons": [],
   "summary": "1,899 tasks testing whether a model can learn new domain knowledge, rules or procedures from its own prompt and apply them; the best of ten frontier models solved only 23.7%."
  },
  {
   "aliases": [
    "CLinical Entity Augmented Retrieval",
    "MedHELM CLEAR"
   ],
   "canonical_id": "clear",
   "category": "domain",
   "disposition": "unassessed",
   "id": "clear",
   "models_covered": 0,
   "name": "CLEAR (MedHELM)",
   "reasons": [],
   "summary": "MedHELM three-way classification of whether a clinical note supports, denies, or is uncertain about a patient's history of one of 13 conditions."
  },
  {
   "aliases": [
    "CLEVA: Chinese Language Models EVAluation Platform",
    "Chinese Language Models EVAluation Platform"
   ],
   "canonical_id": "cleva",
   "category": "composite",
   "disposition": "unassessed",
   "id": "cleva",
   "models_covered": 0,
   "name": "CLEVA",
   "reasons": [],
   "summary": "HELM wrap of CLEVA, a 31-task Chinese LLM platform with standardized prompts, 370K test instances, and contamination-aware sampling."
  },
  {
   "aliases": [
    "CLIcK: Cultural and Linguistic Intelligence in Korean",
    "Cultural and Linguistic Intelligence in Korean"
   ],
   "canonical_id": "click",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "click",
   "models_covered": 0,
   "name": "CLIcK",
   "reasons": [],
   "summary": "Korean multiple-choice exam of cultural and linguistic knowledge: 1,995 questions in eleven categories, drawn from official exams and textbooks."
  },
  {
   "aliases": [],
   "canonical_id": "climaqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "climaqa",
   "models_covered": 0,
   "name": "ClimaQA",
   "reasons": [],
   "summary": "An automated framework producing 566 expert-verified (Gold) and 3,000 synthetic (Silver) climate-science QA items across multiple-choice, cloze and free-form formats."
  },
  {
   "aliases": [
    "Pharmacology QA for Emerging Drugs"
   ],
   "canonical_id": "clinicbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "clinicbench",
   "models_covered": 0,
   "name": "ClinicBench",
   "reasons": [],
   "summary": "OpenCompass's ClinicBench task runs only a 213-item pharmacology-QA slice of a much larger, unrelated-looking 17-dataset clinical benchmark suite published under the same name."
  },
  {
   "aliases": [
    "clozeTest_maxmin",
    "CodeXGLUE ClozeTest-maxmin",
    "maxmin"
   ],
   "canonical_id": "clozetest_maxmin",
   "category": "coding",
   "disposition": "unassessed",
   "id": "clozetest_maxmin",
   "models_covered": 0,
   "name": "ClozeTest-maxmin",
   "reasons": [],
   "summary": "OpenCompass A/B cloze on CodeXGLUE ClozeTest-maxmin: given masked code and a docstring, choose whether the blank is max or min."
  },
  {
   "aliases": [
    "Chinese Language Understanding Evaluation Benchmark",
    "ChineseGLUE"
   ],
   "canonical_id": "clue",
   "category": "composite",
   "disposition": "unassessed",
   "id": "clue",
   "models_covered": 0,
   "name": "CLUE (Chinese Language Understanding Evaluation)",
   "reasons": [],
   "summary": "A nine-task Chinese counterpart to GLUE/SuperGLUE spanning classification, NLI and reading comprehension; its own composite leaderboard has matched or beaten its human baseline since 2023."
  },
  {
   "aliases": [],
   "canonical_id": "clue_afqmc",
   "category": "composite",
   "disposition": "unassessed",
   "id": "clue_afqmc",
   "models_covered": 0,
   "name": "CLUE: AFQMC (Ant Financial Question Matching Corpus)",
   "reasons": [],
   "summary": "CLUE's binary paraphrase task: do two short Chinese questions from Ant Financial's customer-service logs mean the same thing."
  },
  {
   "aliases": [],
   "canonical_id": "clue_c3",
   "category": "composite",
   "disposition": "unassessed",
   "id": "clue_c3",
   "models_covered": 0,
   "name": "CLUE: C3 (free-form multiple-choice Chinese reading comprehension)",
   "reasons": [],
   "summary": "CLUE's multiple-choice reading task, adopted from the separately published C3 dataset spanning Chinese dialogue and mixed-genre text, 2-4 options per question."
  },
  {
   "aliases": [],
   "canonical_id": "clue_cmnli",
   "category": "composite",
   "disposition": "unassessed",
   "id": "clue_cmnli",
   "models_covered": 0,
   "name": "CLUE: CMNLI (Chinese Multi-Genre NLI)",
   "reasons": [],
   "summary": "CLUE's translated NLI task, built from machine-translated MultiNLI and XNLI merged into one set; formally replaced by OCNLI on CLUE's own leaderboard."
  },
  {
   "aliases": [],
   "canonical_id": "clue_cmrc",
   "category": "composite",
   "disposition": "unassessed",
   "id": "clue_cmrc",
   "models_covered": 0,
   "name": "CLUE: CMRC 2018 (Simplified Chinese span-extraction reading comprehension)",
   "reasons": [],
   "summary": "CLUE's simplified-Chinese span-extraction reading task, adopted wholesale from HFL's separately published CMRC 2018 shared-task dataset."
  },
  {
   "aliases": [],
   "canonical_id": "clue_drcd",
   "category": "composite",
   "disposition": "unassessed",
   "id": "clue_drcd",
   "models_covered": 0,
   "name": "CLUE: DRCD (Traditional Chinese span-extraction reading comprehension)",
   "reasons": [],
   "summary": "CLUE's Traditional-Chinese span-extraction task, adopted unchanged from the separately published Delta Reading Comprehension Dataset; not scored on CLUE's live leaderboard."
  },
  {
   "aliases": [],
   "canonical_id": "clue_ocnli",
   "category": "composite",
   "disposition": "unassessed",
   "id": "clue_ocnli",
   "models_covered": 0,
   "name": "CLUE: OCNLI (Original Chinese Natural Language Inference)",
   "reasons": [],
   "summary": "CLUE's native-Chinese NLI task, collected without translation; replaced CMNLI as the suite's scored natural-language-inference task from the 1.1 leaderboard onward."
  },
  {
   "aliases": [
    "Comprehensive Medical Benchmark in Chinese"
   ],
   "canonical_id": "cmb",
   "category": "domain",
   "disposition": "unassessed",
   "id": "cmb",
   "models_covered": 0,
   "name": "CMB (Comprehensive Medical Benchmark in Chinese)",
   "reasons": [],
   "summary": "A two-part Chinese medical benchmark: 280,839 licensing-exam questions across 28 subcategories (CMB-Exam) plus 74 multi-turn clinical cases graded on four qualitative dimensions (CMB-Clin)."
  },
  {
   "aliases": [
    "Chinese Massive Multitask Language Understanding"
   ],
   "canonical_id": "cmmlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "cmmlu",
   "models_covered": 0,
   "name": "CMMLU (Chinese Massive Multitask Language Understanding)",
   "reasons": [],
   "summary": "11,528 multiple-choice questions across 67 subjects, natively authored in Chinese rather than translated, including China-specific subjects such as driving rules and Chinese civil-service topics."
  },
  {
   "aliases": [
    "cmo_fib",
    "Chinese Mathematical Olympiad FIB",
    "CMO FIB"
   ],
   "canonical_id": "cmo_fib",
   "category": "math",
   "disposition": "unassessed",
   "id": "cmo_fib",
   "models_covered": 0,
   "name": "CMO fill-in-the-blank",
   "reasons": [],
   "summary": "OpenCompass fill-in-the-blank set of Chinese Mathematical Olympiad problems from 2009\u20132022, scored with a MATH-style boxed-answer matcher."
  },
  {
   "aliases": [],
   "canonical_id": "cmphysbench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "cmphysbench",
   "models_covered": 0,
   "name": "CMPhysBench",
   "reasons": [],
   "summary": "520 graduate-level condensed matter physics calculation problems scored by a symbolic partial-credit metric; the best model reached only 36 average SEED score and 28% accuracy."
  },
  {
   "aliases": [
    "cnn_dailymail",
    "CNN/DailyMail",
    "CNN-DM",
    "abisee/cnn_dailymail"
   ],
   "canonical_id": "cnn_dailymail_abisee",
   "category": "generation",
   "disposition": "unassessed",
   "id": "cnn_dailymail_abisee",
   "models_covered": 0,
   "name": "CNN/DailyMail (lm-eval, See et al. v3.0.0)",
   "reasons": [],
   "summary": "lm-eval zero-shot abstractive summarization of CNN/DailyMail articles (See et al. version 3.0.0), scored with ROUGE-1/2/L and BERTScore."
  },
  {
   "aliases": [],
   "canonical_id": "coco_bench",
   "category": "coding",
   "disposition": "unassessed",
   "id": "coco_bench",
   "models_covered": 0,
   "name": "CoCo-Bench",
   "reasons": [],
   "summary": "CoCo-Bench evaluates language models across code understanding, generation, modification and review."
  },
  {
   "aliases": [
    "Coconot",
    "Contextually, Comply Not",
    "The Art of Saying No"
   ],
   "canonical_id": "coconot",
   "category": "safety",
   "disposition": "unassessed",
   "id": "coconot",
   "models_covered": 0,
   "name": "CoCoNot",
   "reasons": [],
   "summary": "Allen AI CoCoNot scores whether a chat model withholds answers on 1,001 contextual noncompliance prompts, plus a 379-item contrast set for over-refusal."
  },
  {
   "aliases": [
    "code line description",
    "Code Description"
   ],
   "canonical_id": "code_line_description",
   "category": "coding",
   "disposition": "unassessed",
   "id": "code_line_description",
   "models_covered": 0,
   "name": "Code Line Description",
   "reasons": [],
   "summary": "A 60-item BIG-bench Lite multiple-choice task that asks which English sentence correctly describes a short Python snippet."
  },
  {
   "aliases": [
    "Code-X-GLUE",
    "Code X GLUE",
    "codexglue",
    "code2text"
   ],
   "canonical_id": "code_x_glue",
   "category": "coding",
   "disposition": "unassessed",
   "id": "code_x_glue",
   "models_covered": 0,
   "name": "CodeXGLUE",
   "reasons": [],
   "summary": "Microsoft CodeXGLUE is a 10-task, 14-dataset suite for code understanding and generation; EleutherAI lm-eval currently ships only the code-to-text group."
  },
  {
   "aliases": [
    "CodeComPass",
    "codecompass_gen_cpp"
   ],
   "canonical_id": "codecompass",
   "category": "coding",
   "disposition": "unassessed",
   "id": "codecompass",
   "models_covered": 0,
   "name": "CodeCompass",
   "reasons": [],
   "summary": "OpenCompass C++ pass@1 on CodeCompass: 270 recent AtCoder, Codeforces and Nowcoder problems with SAGA-generated tests."
  },
  {
   "aliases": [
    "CodeInsightsCodeEfficiencyScenario",
    "codeinsights code efficiency"
   ],
   "canonical_id": "codeinsights_code_efficiency",
   "category": "coding",
   "disposition": "unassessed",
   "id": "codeinsights_code_efficiency",
   "models_covered": 0,
   "name": "CodeInsights Code Efficiency",
   "reasons": [],
   "summary": "HELM scenario that asks a model to write C++ matching one student's runtime style, then compares wall-clock times on unit tests."
  },
  {
   "aliases": [
    "CodeInsightsCorrectCodeScenario",
    "codeinsights correct code"
   ],
   "canonical_id": "codeinsights_correct_code",
   "category": "coding",
   "disposition": "unassessed",
   "id": "codeinsights_correct_code",
   "models_covered": 0,
   "name": "CodeInsights Correct Code",
   "reasons": [],
   "summary": "HELM scenario that asks a model to fill a C++ course template so the generated body passes the problem's unit tests."
  },
  {
   "aliases": [
    "CodeInsightsEdgeCaseScenario",
    "codeinsights edge case"
   ],
   "canonical_id": "codeinsights_edge_case",
   "category": "coding",
   "disposition": "unassessed",
   "id": "codeinsights_edge_case",
   "models_covered": 0,
   "name": "CodeInsights Edge Case",
   "reasons": [],
   "summary": "HELM scenario that asks a model to name the one unit-test index a given C++ student is most likely to fail."
  },
  {
   "aliases": [
    "CodeInsightsStudentCodingScenario",
    "codeinsights student coding"
   ],
   "canonical_id": "codeinsights_student_coding",
   "category": "coding",
   "disposition": "unassessed",
   "id": "codeinsights_student_coding",
   "models_covered": 0,
   "name": "CodeInsights Student Coding",
   "reasons": [],
   "summary": "HELM scenario that asks a model to write C++ in one student's style given three of their submissions and a new problem."
  },
  {
   "aliases": [
    "CodeInsightsStudentMistakeScenario",
    "codeinsights student mistake"
   ],
   "canonical_id": "codeinsights_student_mistake",
   "category": "coding",
   "disposition": "unassessed",
   "id": "codeinsights_student_mistake",
   "models_covered": 0,
   "name": "CodeInsights Student Mistake",
   "reasons": [],
   "summary": "HELM scenario that asks a model to introduce a given student's typical C++ mistakes on a new problem, using three prior buggy submissions."
  },
  {
   "aliases": [
    "BIG-bench codenames"
   ],
   "canonical_id": "codenames",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "codenames",
   "models_covered": 0,
   "name": "Codenames",
   "reasons": [],
   "summary": "An 85-item BIG-bench free-response task: given a Codenames-style clue and a word list, emit the associated words in alphabetical order."
  },
  {
   "aliases": [
    "BIG-bench color"
   ],
   "canonical_id": "color",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "color",
   "models_covered": 0,
   "name": "Color",
   "reasons": [],
   "summary": "A 4,000-item BIG-bench task that maps RGB, HEX, HSL, and HCL encodings to one of ten English color names."
  },
  {
   "aliases": [
    "COM2SENSE",
    "BIG-bench com2sense"
   ],
   "canonical_id": "com2sense",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "com2sense",
   "models_covered": 0,
   "name": "Com2Sense",
   "reasons": [],
   "summary": "A BIG-bench programmatic slice of Com2Sense: complementary true/false sentences scored with pairwise accuracy."
  },
  {
   "aliases": [
    "BIG-bench common_morpheme"
   ],
   "canonical_id": "common_morpheme",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "common_morpheme",
   "models_covered": 0,
   "name": "Common Morpheme",
   "reasons": [],
   "summary": "A 50-item BIG-bench 4-way task (25 English, 25 French) that asks for the shared morpheme meaning across a word list."
  },
  {
   "aliases": [
    "commonsenseqa"
   ],
   "canonical_id": "commonsense_qa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "commonsense_qa",
   "models_covered": 0,
   "name": "CommonsenseQA",
   "reasons": [],
   "summary": "A 5-way multiple-choice commonsense test built from ConceptNet relations, designed so questions need world knowledge beyond the immediate text."
  },
  {
   "aliases": [
    "commonsenseqacn",
    "CommonSenseQA-CN"
   ],
   "canonical_id": "commonsenseqa_cn",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "commonsenseqa_cn",
   "models_covered": 0,
   "name": "CommonsenseQA-CN",
   "reasons": [],
   "summary": "OpenCompass's Chinese-prompt 5-way CommonsenseQA wrap, scored as accuracy on a local validation.jsonl file."
  },
  {
   "aliases": [
    "CompassBench 2.0 v1.1",
    "compassbench_20_v1_1"
   ],
   "canonical_id": "compassbench_20_v1_1",
   "category": "composite",
   "disposition": "unassessed",
   "id": "compassbench_20_v1_1",
   "models_covered": 0,
   "name": "CompassBench v1.1",
   "reasons": [],
   "summary": "OpenCompass's own composite: naive-averages six category scores (language, knowledge, reasoning, math, code, agent), each itself an average across many pre-existing and self-built datasets."
  },
  {
   "aliases": [
    "compassbench_20_v1_1_public"
   ],
   "canonical_id": "compassbench_20_v1_1_public",
   "category": "composite",
   "disposition": "unassessed",
   "id": "compassbench_20_v1_1_public",
   "models_covered": 0,
   "name": "CompassBench v1.1 (public)",
   "reasons": [],
   "summary": "The publicly redistributed, reduced-item release of CompassBench v1.1: the same six-category structure, with some code tasks restricted to their first five test cases."
  },
  {
   "aliases": [
    "compassbench_v1_3"
   ],
   "canonical_id": "compassbench_v1_3",
   "category": "composite",
   "disposition": "unassessed",
   "id": "compassbench_v1_3",
   "models_covered": 0,
   "name": "CompassBench v1.3",
   "reasons": [],
   "summary": "OpenCompass's next dated CompassBench release: a narrower, restructured composite of knowledge, math, code and agent scores, not a superset of CompassBench v1.1."
  },
  {
   "aliases": [
    "NVIDIA ComputeEval",
    "compute-eval"
   ],
   "canonical_id": "compute_eval",
   "category": "coding",
   "disposition": "unassessed",
   "id": "compute_eval",
   "models_covered": 0,
   "name": "ComputeEval",
   "reasons": [],
   "summary": "NVIDIA's benchmark of CUDA programming challenges - a model must write CUDA code that compiles with nvcc and passes a held-out test harness, spanning kernels, runtime APIs and GPU libraries."
  },
  {
   "aliases": [
    "BIG-bench conceptual_combinations"
   ],
   "canonical_id": "conceptual_combinations",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "conceptual_combinations",
   "models_covered": 0,
   "name": "Conceptual Combinations",
   "reasons": [],
   "summary": "A 103-item BIG-bench Lite four-way task that asks which English sentence uses a two-word conceptual combination correctly."
  },
  {
   "aliases": [
    "BIG-bench conlang_translation",
    "Conlang Translation Problems"
   ],
   "canonical_id": "conlang_translation",
   "category": "translation",
   "disposition": "unassessed",
   "id": "conlang_translation",
   "models_covered": 0,
   "name": "Conlang Translation",
   "reasons": [],
   "summary": "A 164-item BIG-bench Lite task that infers a ciphered language from a few examples, then translates to or from English."
  },
  {
   "aliases": [
    "CoDA",
    "BIG-bench context_definition_alignment"
   ],
   "canonical_id": "context_definition_alignment",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "context_definition_alignment",
   "models_covered": 0,
   "name": "Context Definition Alignment",
   "reasons": [],
   "summary": "A programmatic BIG-bench task that matches masked SemCor contexts to WordNet definitions across 3,184 CoDA synset groups."
  },
  {
   "aliases": [
    "BIG-bench contextual_parametric_knowledge_conflicts"
   ],
   "canonical_id": "contextual_parametric_knowledge_conflicts",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "contextual_parametric_knowledge_conflicts",
   "models_covered": 0,
   "name": "Contextual Parametric Knowledge Conflicts",
   "reasons": [],
   "summary": "A 17,528-item BIG-bench QA task that swaps person names in Natural Questions passages so the context contradicts memorized facts."
  },
  {
   "aliases": [
    "conv_fin_qa_calc",
    "HELM ConvFinQACalc"
   ],
   "canonical_id": "conv_fin_qa_calc",
   "category": "domain",
   "disposition": "unassessed",
   "id": "conv_fin_qa_calc",
   "models_covered": 0,
   "name": "ConvFinQACalc",
   "reasons": [],
   "summary": "HELM Enterprise wrap of ConvFinQA: given a table, gold text facts, and prior turns, emit the last-turn number and score float equality."
  },
  {
   "aliases": [
    "BIG-bench convinceme",
    "ConvinceMe"
   ],
   "canonical_id": "convinceme",
   "category": "safety",
   "disposition": "unassessed",
   "id": "convinceme",
   "models_covered": 0,
   "name": "Convince Me",
   "reasons": [],
   "summary": "A programmatic BIG-bench task that scores how far a model sways copies of itself toward 455 true and false TruthfulQA statements."
  },
  {
   "aliases": [
    "COPAL-ID",
    "COPAL",
    "copal_id_standard",
    "copal_id_colloquial"
   ],
   "canonical_id": "copal_id",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "copal_id",
   "models_covered": 0,
   "name": "COPAL-ID (Choice of Plausible Alternatives \u2014 Local Nuances, Indonesia)",
   "reasons": [],
   "summary": "Two 559-item Indonesian COPA-style tests of Jakartan local terms, culture, and language, written from scratch in standard and colloquial Indonesian."
  },
  {
   "aliases": [
    "copyright_text",
    "copyright_code",
    "HELM copyright"
   ],
   "canonical_id": "copyright",
   "category": "safety",
   "disposition": "unassessed",
   "id": "copyright",
   "models_covered": 0,
   "name": "Copyright (HELM memorisation / extraction)",
   "reasons": [],
   "summary": "HELM scenario that feeds a book or Linux-kernel prefix and scores how closely the model continues the copyrighted remainder."
  },
  {
   "aliases": [
    "CoQA"
   ],
   "canonical_id": "coqa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "coqa",
   "models_covered": 0,
   "name": "CoQA (Conversational Question Answering Challenge)",
   "reasons": [],
   "summary": "Free-form question answering over a passage, where the questions form a real conversation and each answer needs the prior turns to be understood."
  },
  {
   "aliases": [
    "Computational Reproducibility Benchmark"
   ],
   "canonical_id": "core_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "core_bench",
   "models_covered": 0,
   "name": "CORE-Bench",
   "reasons": [],
   "summary": "Tests whether an agent can reproduce a published paper's results by installing dependencies, running its code, and extracting the right numbers, across three levels of given scaffolding."
  },
  {
   "aliases": [],
   "canonical_id": "core_bench_computational_reproducibility_agent_benchmark",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "core_bench_computational_reproducibility_agent_benchmark",
   "models_covered": 0,
   "name": "CORE-Bench",
   "reasons": [],
   "summary": "CORE-Bench measures whether agents can reproduce scientific results from provided code and data."
  },
  {
   "aliases": [
    "COVIDDialog",
    "covid_dialogue",
    "CovidDialog"
   ],
   "canonical_id": "covid_dialog",
   "category": "domain",
   "disposition": "unassessed",
   "id": "covid_dialog",
   "models_covered": 0,
   "name": "COVIDDialog (HELM English medical dialogue)",
   "reasons": [],
   "summary": "HELM generation task: read an English COVID-19 patient question and write the doctor's reply, scored with overlap metrics."
  },
  {
   "aliases": [
    "Crash Blossoms",
    "crash blossom"
   ],
   "canonical_id": "crash_blossom",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "crash_blossom",
   "models_covered": 0,
   "name": "Crash Blossoms (BIG-bench)",
   "reasons": [],
   "summary": "A 38-item BIG-bench task that asks for the part of speech of an ambiguous word in a crash-blossom news headline."
  },
  {
   "aliases": [
    "CRASS",
    "crass_ai",
    "Counterfactual Conditionals"
   ],
   "canonical_id": "crass_ai",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "crass_ai",
   "models_covered": 0,
   "name": "CRASS (BIG-bench crass_ai)",
   "reasons": [],
   "summary": "A 44-item BIG-bench multiple-choice slice of CRASS: what would have happened if a simple event had gone the other way."
  },
  {
   "aliases": [
    "CritPt",
    "Critical Point",
    "CritPt_main"
   ],
   "canonical_id": "critpt",
   "category": "reasoning",
   "disposition": "active",
   "id": "critpt",
   "models_covered": 0,
   "name": "CritPt (Complex Research using Integrated Thinking \u2014 Physics Test)",
   "reasons": [
    "qualifying current coverage"
   ],
   "summary": "Seventy public research-physics challenges with held-out answers, graded by a server; full suite is 71 challenges plus 190 checkpoints."
  },
  {
   "aliases": [
    "Crowdsourced Stereotype Pairs",
    "crows-pairs",
    "crows_pairs_english"
   ],
   "canonical_id": "crows_pairs",
   "category": "safety",
   "disposition": "unassessed",
   "id": "crows_pairs",
   "models_covered": 0,
   "name": "CrowS-Pairs",
   "reasons": [],
   "summary": "Crowdsourced English sentence pairs that score whether a language model assigns higher likelihood to the more stereotyping sentence than to its minimally edited counterpart."
  },
  {
   "aliases": [
    "crowspairscn",
    "CrowspairsDatasetCN"
   ],
   "canonical_id": "crowspairs_cn",
   "category": "safety",
   "disposition": "unassessed",
   "id": "crowspairs_cn",
   "models_covered": 0,
   "name": "CrowS-Pairs-CN",
   "reasons": [],
   "summary": "OpenCompass's Chinese CrowS-Pairs wrap: the model must pick the less-biased sentence of a pair, scored as accuracy under perplexity or generative A/B prompts."
  },
  {
   "aliases": [
    "CRUXEval: Code Reasoning, Understanding, and Execution Evaluation"
   ],
   "canonical_id": "cruxeval",
   "category": "coding",
   "disposition": "unassessed",
   "id": "cruxeval",
   "models_covered": 0,
   "name": "CRUXEval",
   "reasons": [],
   "summary": "Given a short Python function and either its input or its output, predict the other by reasoning about execution -- not by writing new code."
  },
  {
   "aliases": [
    "cryobiology spanish"
   ],
   "canonical_id": "cryobiology_spanish",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "cryobiology_spanish",
   "models_covered": 0,
   "name": "Cryobiology Spanish",
   "reasons": [],
   "summary": "A 146-item Spanish two-way BIG-bench quiz on cryobiology facts, scored by multiple-choice grade."
  },
  {
   "aliases": [
    "cryptic crossword"
   ],
   "canonical_id": "cryptonite",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "cryptonite",
   "models_covered": 0,
   "name": "Cryptonite",
   "reasons": [],
   "summary": "BIG-bench wrap of Cryptonite: 26,157 free-text cryptic crossword clues scored by exact string match."
  },
  {
   "aliases": [
    "cs algorithms"
   ],
   "canonical_id": "cs_algorithms",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "cs_algorithms",
   "models_covered": 0,
   "name": "CS Algorithms",
   "reasons": [],
   "summary": "A 1,320-item BIG-bench pair of multiple-choice algorithm tasks: balanced parentheses and longest common subsequence."
  },
  {
   "aliases": [
    "CSAT-QA(EVAL)"
   ],
   "canonical_id": "csatqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "csatqa",
   "models_covered": 0,
   "name": "CSAT-QA",
   "reasons": [],
   "summary": "187 Korean CSAT exam questions across six subjects, each with a recorded human student accuracy; GPT-4 already beat the average human score in the original 2023 evaluation."
  },
  {
   "aliases": [
    "CTI REALM",
    "Cyber Threat Real World Evaluation and LLM Benchmarking"
   ],
   "canonical_id": "cti_realm",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "cti_realm",
   "models_covered": 0,
   "name": "CTI-REALM",
   "reasons": [],
   "summary": "Inspect AI agent benchmark where a model writes Sigma and KQL detection rules from cyber threat intelligence reports against live Kusto telemetry."
  },
  {
   "aliases": [
    "CTI to MITRE",
    "cti-to-mitre-with-nlp"
   ],
   "canonical_id": "cti_to_mitre",
   "category": "domain",
   "disposition": "unassessed",
   "id": "cti_to_mitre",
   "models_covered": 0,
   "name": "CTI-to-MITRE",
   "reasons": [],
   "summary": "HELM multiple-choice wrap that maps a short cyber threat intelligence sentence to a MITRE ATT&CK enterprise technique name."
  },
  {
   "aliases": [
    "CValues-Responsibility",
    "CVALUES"
   ],
   "canonical_id": "cvalues",
   "category": "safety",
   "disposition": "unassessed",
   "id": "cvalues",
   "models_covered": 0,
   "name": "CValues",
   "reasons": [],
   "summary": "Chinese two-choice value-alignment eval whose public OpenCompass slice is 1,712 responsibility items that ask which of two replies is more responsible."
  },
  {
   "aliases": [
    "CVE-Bench"
   ],
   "canonical_id": "cve_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "cve_bench",
   "models_covered": 0,
   "name": "CVE-Bench",
   "reasons": [],
   "summary": "Tests whether an autonomous LLM agent can actually exploit 40 real, critical-severity web application CVEs inside a live sandbox, not whether it can answer security questions."
  },
  {
   "aliases": [
    "Cybench: A Framework for Evaluating Cybersecurity Capabilities and Risks of Language Models"
   ],
   "canonical_id": "cybench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "cybench",
   "models_covered": 0,
   "name": "Cybench",
   "reasons": [],
   "summary": "Tests whether an autonomous LLM agent can solve 40 professional-level capture-the-flag cybersecurity tasks in a live sandbox, with 17 of them broken into guided subtasks for partial credit."
  },
  {
   "aliases": [
    "CyberGym: Evaluating AI Agents' Real-World Cybersecurity Capabilities at Scale"
   ],
   "canonical_id": "cybergym",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "cybergym",
   "models_covered": 0,
   "name": "CyberGym",
   "reasons": [],
   "summary": "1,507 real patched vulnerabilities across 188 projects; agents must write a PoC that crashes the unpatched binary but not the fix."
  },
  {
   "aliases": [
    "CyberMetric-80",
    "CyberMetric-500",
    "CyberMetric-2000",
    "CyberMetric-10000"
   ],
   "canonical_id": "cybermetric",
   "category": "domain",
   "disposition": "unassessed",
   "id": "cybermetric",
   "models_covered": 0,
   "name": "CyberMetric",
   "reasons": [],
   "summary": "Four public cybersecurity multiple-choice sets (80, 500, 2,000, 10,000 items) built with RAG and expert review to test LLM security knowledge."
  },
  {
   "aliases": [
    "CYBERSECEVAL 2",
    "Purple Llama CyberSecEval 2"
   ],
   "canonical_id": "cyberseceval_2",
   "category": "safety",
   "disposition": "unassessed",
   "id": "cyberseceval_2",
   "models_covered": 0,
   "name": "CyberSecEval 2",
   "reasons": [],
   "summary": "Meta's second-generation LLM cybersecurity suite: adds prompt-injection, code-interpreter-abuse and exploit-capability tests to the original CyberSecEval's insecure-code and attack-compliance measures."
  },
  {
   "aliases": [
    "CYBERSECEVAL 3",
    "Purple Llama CyberSecEval 3"
   ],
   "canonical_id": "cyberseceval_3",
   "category": "safety",
   "disposition": "unassessed",
   "id": "cyberseceval_3",
   "models_covered": 0,
   "name": "CyberSecEval 3",
   "reasons": [],
   "summary": "Meta's third CyberSecEval release: adds visual prompt injection, automated spear-phishing, and human and autonomous offensive-cyber-operations studies to the prior version's risk and insecure-code tests."
  },
  {
   "aliases": [
    "CYBERSECEVAL 4",
    "Purple Llama CyberSecEval 4"
   ],
   "canonical_id": "cyberseceval_4",
   "category": "safety",
   "disposition": "unassessed",
   "id": "cyberseceval_4",
   "models_covered": 0,
   "name": "CyberSecEval 4",
   "reasons": [],
   "summary": "Meta's fourth CyberSecEval release, adding CrowdStrike-built defensive benchmarks (malware analysis, threat-intel reasoning) and an automated-patching benchmark to the prior risk and insecure-code tests."
  },
  {
   "aliases": [
    "cycled letters"
   ],
   "canonical_id": "cycled_letters",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "cycled_letters",
   "models_covered": 0,
   "name": "Cycled Letters (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench programmatic task: rotate a filtered English word and require the original spelling by exact string match."
  },
  {
   "aliases": [],
   "canonical_id": "czech_bank_qa",
   "category": "coding",
   "disposition": "unassessed",
   "id": "czech_bank_qa",
   "models_covered": 0,
   "name": "CzechBankQA",
   "reasons": [],
   "summary": "An experimental HELM text-to-SQL scenario that asks a model to write a SQLite query, in English, over the classic 1999 Czech Bank relational dataset, graded only on whether the query runs."
  },
  {
   "aliases": [],
   "canonical_id": "darija_bench",
   "category": "composite",
   "disposition": "unassessed",
   "id": "darija_bench",
   "models_covered": 0,
   "name": "DarijaBench",
   "reasons": [],
   "summary": "An 11-task, four-way suite of held-out test slices for Moroccan Darija covering sentiment analysis, summarization, translation and transliteration, built for the Atlas-Chat model release."
  },
  {
   "aliases": [
    "Darija HellaSwag",
    "MBZUAI-Paris/DarijaHellaSwag"
   ],
   "canonical_id": "darijahellaswag",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "darijahellaswag",
   "models_covered": 0,
   "name": "DarijaHellaSwag",
   "reasons": [],
   "summary": "Darija translation of HellaSwag: pick the plausible ending among four Moroccan Arabic continuations of a short scene."
  },
  {
   "aliases": [
    "Darija MMLU",
    "MBZUAI-Paris/DarijaMMLU"
   ],
   "canonical_id": "darijammlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "darijammlu",
   "models_covered": 0,
   "name": "DarijaMMLU",
   "reasons": [],
   "summary": "22,027 Darija multiple-choice questions across 44 subjects, translated from selected MMLU and ArabicMMLU subsets."
  },
  {
   "aliases": [
    "Dark Humor Detection"
   ],
   "canonical_id": "dark_humor_detection",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "dark_humor_detection",
   "models_covered": 0,
   "name": "Dark Humor Detection (BIG-bench)",
   "reasons": [],
   "summary": "80-item BIG-bench binary task: decide whether a short English text is intended as a dark joke."
  },
  {
   "aliases": [
    "Date Understanding",
    "BBH date_understanding"
   ],
   "canonical_id": "date_understanding",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "date_understanding",
   "models_covered": 0,
   "name": "Date Understanding",
   "reasons": [],
   "summary": "A 369-item BIG-bench multiple-choice task: infer a calendar date in MM/DD/YYYY from one or two English sentences."
  },
  {
   "aliases": [
    "AdvDemo",
    "decodingtrust_adv_demo",
    "DecodingTrustAdvDemoScenario"
   ],
   "canonical_id": "decodingtrust_adv_demonstration",
   "category": "safety",
   "disposition": "unassessed",
   "id": "decodingtrust_adv_demonstration",
   "models_covered": 0,
   "name": "DecodingTrust Adversarial Demonstrations",
   "reasons": [],
   "summary": "HELM scenario for DecodingTrust's adversarial-demonstration tests: counterfactual, spurious, and backdoored in-context examples."
  },
  {
   "aliases": [
    "AdvGLUE++",
    "decodingtrust_adv_glue_plus_plus",
    "DecodingTrustAdvRobustnessScenario"
   ],
   "canonical_id": "decodingtrust_adv_robustness",
   "category": "safety",
   "disposition": "unassessed",
   "id": "decodingtrust_adv_robustness",
   "models_covered": 0,
   "name": "DecodingTrust Adversarial Robustness (AdvGLUE++)",
   "reasons": [],
   "summary": "HELM scenario for DecodingTrust AdvGLUE++: GLUE items rewritten against Alpaca, Vicuna, and StableVicuna."
  },
  {
   "aliases": [
    "DecodingTrustFairnessScenario",
    "DecodingTrust - Fairness"
   ],
   "canonical_id": "decodingtrust_fairness",
   "category": "safety",
   "disposition": "unassessed",
   "id": "decodingtrust_fairness",
   "models_covered": 0,
   "name": "DecodingTrust Fairness",
   "reasons": [],
   "summary": "HELM scenario for DecodingTrust fairness: yes/no Adult income prediction, scored for accuracy and group gaps."
  },
  {
   "aliases": [
    "DecodingTrust - Ethics",
    "DecodingTrustMachineEthicsScenario"
   ],
   "canonical_id": "decodingtrust_machine_ethics",
   "category": "safety",
   "disposition": "unassessed",
   "id": "decodingtrust_machine_ethics",
   "models_covered": 0,
   "name": "DecodingTrust Machine Ethics",
   "reasons": [],
   "summary": "HELM scenario for DecodingTrust ethics: wrong/not-wrong labels on ETHICS and Jiminy Cricket, with jailbreak and evasive filters."
  },
  {
   "aliases": [
    "DecodingTrust - OoD Robustness",
    "DecodingTrustOODRobustnessScenario",
    "decodingtrust_ood"
   ],
   "canonical_id": "decodingtrust_ood_robustness",
   "category": "safety",
   "disposition": "unassessed",
   "id": "decodingtrust_ood_robustness",
   "models_covered": 0,
   "name": "DecodingTrust OoD Robustness",
   "reasons": [],
   "summary": "HELM scenario for DecodingTrust out-of-distribution robustness: style-shifted SST-2 sentiment and RealtimeQA knowledge with optional refusal."
  },
  {
   "aliases": [
    "DecodingTrust - Privacy"
   ],
   "canonical_id": "decodingtrust_privacy",
   "category": "safety",
   "disposition": "unassessed",
   "id": "decodingtrust_privacy",
   "models_covered": 0,
   "name": "DecodingTrust Privacy",
   "reasons": [],
   "summary": "DecodingTrust's privacy slice: extract Enron emails, leak planted PII, or share a 'secret' after a privacy cue."
  },
  {
   "aliases": [
    "DecodingTrust - Stereotype Bias",
    "DecodingTrustStereotypeBiasScenario"
   ],
   "canonical_id": "decodingtrust_stereotype_bias",
   "category": "safety",
   "disposition": "unassessed",
   "id": "decodingtrust_stereotype_bias",
   "models_covered": 0,
   "name": "DecodingTrust Stereotype Bias",
   "reasons": [],
   "summary": "HELM scenario for DecodingTrust stereotypes: agree/disagree with 1,152 statements across 16 topics, 24 groups, and 3 system prompts."
  },
  {
   "aliases": [
    "DecodingTrust - Toxicity",
    "DecodingTrustToxicityPromptsScenario"
   ],
   "canonical_id": "decodingtrust_toxicity_prompts",
   "category": "safety",
   "disposition": "unassessed",
   "id": "decodingtrust_toxicity_prompts",
   "models_covered": 0,
   "name": "DecodingTrust Toxicity Prompts",
   "reasons": [],
   "summary": "HELM scenario for DecodingTrust toxicity: continue RealToxicityPrompts-style prefixes and score PerspectiveAPI toxic_frac."
  },
  {
   "aliases": [
    "MRCR v2",
    "multi-round coreference resolution",
    "DeepMindMRCRV2Scenario",
    "mrcr_v2p1"
   ],
   "canonical_id": "deepmind_mrcr_v2",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "deepmind_mrcr_v2",
   "models_covered": 0,
   "name": "DeepMind MRCR v2",
   "reasons": [],
   "summary": "Google DeepMind's MRCR v2: reproduce the i-th matching assistant turn in a long dialogue, after emitting a 12-character hash."
  },
  {
   "aliases": [],
   "canonical_id": "deepsearchqa",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "deepsearchqa",
   "models_covered": 1,
   "name": "DeepSearchQA",
   "reasons": [],
   "summary": "900 multi-step web research tasks that grade an agent's full, deduplicated answer set rather than one fact."
  },
  {
   "aliases": [
    "dingo-python",
    "DingoDataset",
    "DingoEvaluator"
   ],
   "canonical_id": "dingo",
   "category": "generation",
   "disposition": "unassessed",
   "id": "dingo",
   "models_covered": 0,
   "name": "Dingo (OpenCompass wrap)",
   "reasons": [],
   "summary": "OpenCompass dataset id dingo: generate from local English/Chinese CSVs, then score completions with dingo-python llm_base rules."
  },
  {
   "aliases": [
    "Disambiguation_QA"
   ],
   "canonical_id": "disambiguation_qa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "disambiguation_qa",
   "models_covered": 0,
   "name": "Disambiguation QA",
   "reasons": [],
   "summary": "A 258-item BIG-bench task that asks a model to resolve an ambiguous pronoun to a person, or say the sentence is ambiguous, testing coreference resolution and gender bias together."
  },
  {
   "aliases": [
    "Discharge Me!",
    "DischargeMe",
    "BioNLP ACL'24 Discharge Me"
   ],
   "canonical_id": "dischargeme",
   "category": "domain",
   "disposition": "unassessed",
   "id": "dischargeme",
   "models_covered": 0,
   "name": "DischargeMe (MedHELM)",
   "reasons": [],
   "summary": "MedHELM's gated wrap of the BioNLP 2024 DischargeMe shared task: write Brief Hospital Course and Discharge Instructions from MIMIC-IV notes and radiology text."
  },
  {
   "aliases": [
    "BIG-bench discourse_marker_prediction"
   ],
   "canonical_id": "discourse_marker_prediction",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "discourse_marker_prediction",
   "models_covered": 0,
   "name": "Discourse Marker Prediction",
   "reasons": [],
   "summary": "An 857-item BIG-bench 10-way task that asks which English discourse marker continues a sentence pair, drawn from Discovery."
  },
  {
   "aliases": [
    "discrim-eval",
    "Evaluating and Mitigating Discrimination in Language Model Decisions"
   ],
   "canonical_id": "discrim_eval",
   "category": "safety",
   "disposition": "unassessed",
   "id": "discrim_eval",
   "models_covered": 0,
   "name": "Discrim-Eval",
   "reasons": [],
   "summary": "Anthropic's test of whether a model's yes/no decisions on 70 high-stakes scenarios shift with a subject's age, gender or race; it produces a discrimination score where closer to zero is better, not a correctness score."
  },
  {
   "aliases": [
    "Disfl-QA",
    "DISFL-QA: A Benchmark Dataset for Understanding Disfluencies in Question Answering"
   ],
   "canonical_id": "disfl_qa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "disfl_qa",
   "models_covered": 0,
   "name": "DISFL-QA",
   "reasons": [],
   "summary": "An 8,000-item BIG-bench task built by inserting realistic speech disfluencies into SQuAD v2 questions, to test whether models stay robust when a question corrects or restarts itself."
  },
  {
   "aliases": [
    "disinformation_reiteration",
    "disinformation_wedging",
    "HELM disinformation"
   ],
   "canonical_id": "disinformation",
   "category": "safety",
   "disposition": "unassessed",
   "id": "disinformation",
   "models_covered": 0,
   "name": "Disinformation (HELM)",
   "reasons": [],
   "summary": "HELM harms scenario that asks a model to write thesis-supporting headlines or group-targeted wedge copy, scored by diversity metrics and optional human ratings."
  },
  {
   "aliases": [
    "Diverse Metrics for Social Biases in Language Models",
    "BIG-bench diverse_social_bias"
   ],
   "canonical_id": "diverse_social_bias",
   "category": "safety",
   "disposition": "unassessed",
   "id": "diverse_social_bias",
   "models_covered": 0,
   "name": "Diverse Social Bias",
   "reasons": [],
   "summary": "A programmatic BIG-bench task that scores gender-occupation local and global bias on naturally occurring English contexts, then negates the scores so higher is fairer."
  },
  {
   "aliases": [
    "Document Visual Question Answering"
   ],
   "canonical_id": "docvqa",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "docvqa",
   "models_covered": 67,
   "name": "DocVQA",
   "reasons": [],
   "summary": "Question answering over scanned and typed document images, scored by fuzzy text match against reference answers."
  },
  {
   "aliases": [],
   "canonical_id": "dr_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "dr_bench",
   "models_covered": 0,
   "name": "Dr. Bench",
   "reasons": [],
   "summary": "Dr. Bench evaluates deep research agents that decompose tasks, retrieve sources, reason across them and produce structured long reports."
  },
  {
   "aliases": [],
   "canonical_id": "dr_docbench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "dr_docbench",
   "models_covered": 0,
   "name": "Dr. DocBench",
   "reasons": [],
   "summary": "Dr. DocBench evaluates expert-level document parsing on difficult multilingual pages with layout, reading-order, and domain-specific annotations."
  },
  {
   "aliases": [
    "DROP: A Reading Comprehension Benchmark Requiring Discrete Reasoning Over Paragraphs"
   ],
   "canonical_id": "drop",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "drop",
   "models_covered": 0,
   "name": "DROP",
   "reasons": [],
   "summary": "An adversarially crowdsourced reading-comprehension test requiring numerical and discrete operations -- addition, counting, sorting -- over a paragraph, not just span lookup."
  },
  {
   "aliases": [],
   "canonical_id": "ds1000",
   "category": "coding",
   "disposition": "unassessed",
   "id": "ds1000",
   "models_covered": 0,
   "name": "DS-1000",
   "reasons": [],
   "summary": "1,000 data-science coding problems across seven Python libraries, perturbed from real StackOverflow questions and checked by execution plus surface-form API constraints."
  },
  {
   "aliases": [
    "Dyck",
    "HELM Dyck"
   ],
   "canonical_id": "dyck_language",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "dyck_language",
   "models_covered": 0,
   "name": "Dyck Language (HELM)",
   "reasons": [],
   "summary": "HELM scenario that generates Dyck-n prefixes at run time and scores the unique closing-bracket suffix with exact match."
  },
  {
   "aliases": [
    "Dyck Language"
   ],
   "canonical_id": "dyck_languages",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "dyck_languages",
   "models_covered": 0,
   "name": "Dyck Languages (BIG-bench)",
   "reasons": [],
   "summary": "1,000 BIG-bench Dyck-4 prefixes; the model must pick the unique closing-bracket suffix from 84 choices."
  },
  {
   "aliases": [
    "Shuffle-n last parenthesis",
    "BIG-bench dynamic_counting"
   ],
   "canonical_id": "dynamic_counting",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "dynamic_counting",
   "models_covered": 0,
   "name": "Dynamic Counting",
   "reasons": [],
   "summary": "A 250-trial BIG-bench task that asks which of four closing brackets finishes a Shuffle-4 prefix, testing counter-style counting rather than full Dyck hierarchy."
  },
  {
   "aliases": [
    "Earth_Silver",
    "earth_silver_mcq",
    "EarthSE Silver"
   ],
   "canonical_id": "earth_silver",
   "category": "domain",
   "disposition": "unassessed",
   "id": "earth_silver",
   "models_covered": 0,
   "name": "Earth-Silver",
   "reasons": [],
   "summary": "EarthSE's harder 1,000-item Earth-science QA split; OpenCompass scores the 250 multiple-choice items, not Iron, Gold, or the other three formats."
  },
  {
   "aliases": [
    "ECHR binary violation",
    "echr_judgment_classification",
    "Neural Legal Judgment Prediction binary task"
   ],
   "canonical_id": "echr_judgment_classification",
   "category": "domain",
   "disposition": "unassessed",
   "id": "echr_judgment_classification",
   "models_covered": 0,
   "name": "ECHR Judgment Classification (HELM)",
   "reasons": [],
   "summary": "HELM Enterprise binary task: given English ECHR facts, answer Yes or No whether any Convention article was violated, using Chalkidis et al. 2019 data."
  },
  {
   "aliases": [
    "Ever-Evolving Science Exam",
    "EESE-V1",
    "EESE-V2",
    "EESE-V3"
   ],
   "canonical_id": "eese",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "eese",
   "models_covered": 0,
   "name": "EESE (Ever-Evolving Science Exam)",
   "reasons": [],
   "summary": "A resampled science exam of about 500 items drawn from a 100K+ pool; OpenCompass grades model answers with an LLM judge on a 0\u201310 scale."
  },
  {
   "aliases": [
    "EgyHellaswag",
    "UBC-NLP/EgyHellaSwag"
   ],
   "canonical_id": "egyhellaswag",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "egyhellaswag",
   "models_covered": 0,
   "name": "EgyHellaSwag",
   "reasons": [],
   "summary": "Egyptian Arabic four-way sentence completion translated from HellaSwag; lm-eval scores the 10,042-item validation split."
  },
  {
   "aliases": [
    "Egy MMLU",
    "UBC-NLP/EgyMMLU"
   ],
   "canonical_id": "egymmlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "egymmlu",
   "models_covered": 0,
   "name": "EgyMMLU",
   "reasons": [],
   "summary": "Egyptian Arabic multiple-choice exam of 22,027 items across 44 subjects, translated from MMLU and ArabicMMLU."
  },
  {
   "aliases": [
    "EHRSQL",
    "EHR SQL"
   ],
   "canonical_id": "ehr_sql",
   "category": "coding",
   "disposition": "unassessed",
   "id": "ehr_sql",
   "models_covered": 0,
   "name": "EHRSQL (HELM ehr_sql / eICU)",
   "reasons": [],
   "summary": "HELM's eICU-only wrap of EHRSQL: write SQL for hospital questions, including unanswerable ones, and score execution accuracy."
  },
  {
   "aliases": [
    "EHRSHOT",
    "EHRShot"
   ],
   "canonical_id": "ehrshot",
   "category": "domain",
   "disposition": "unassessed",
   "id": "ehrshot",
   "models_covered": 0,
   "name": "EHRSHOT (HELM ehrshot)",
   "reasons": [],
   "summary": "HELM's yes/no wrap of Stanford EHRSHOT: predict a future clinical event from EHR codes, scored by exact match, not the paper's AUROC."
  },
  {
   "aliases": [
    "Math Word Problems with Hints"
   ],
   "canonical_id": "elementary_math_qa",
   "category": "math",
   "disposition": "unassessed",
   "id": "elementary_math_qa",
   "models_covered": 0,
   "name": "Elementary Math QA",
   "reasons": [],
   "summary": "A BIG-bench task of MathQA-derived word problems, offered five ways (no hint, a natural-language hint, a raw operation-sequence hint, or the hints alone) to isolate where models fail."
  },
  {
   "aliases": [
    "BIG-bench emoji_movie"
   ],
   "canonical_id": "emoji_movie",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "emoji_movie",
   "models_covered": 0,
   "name": "Emoji Movie",
   "reasons": [],
   "summary": "A 100-item BIG-bench Lite quiz that asks which popular movie an emoji plot describes, scored mainly by five-way multiple_choice_grade."
  },
  {
   "aliases": [
    "BIG-bench emojis_emotion_prediction",
    "EmoTag1200 BIG-bench wrap"
   ],
   "canonical_id": "emojis_emotion_prediction",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "emojis_emotion_prediction",
   "models_covered": 0,
   "name": "Emojis Emotion Prediction",
   "reasons": [],
   "summary": "A 131-item BIG-bench task that maps an emoji to one of Plutchik's eight emotions using human association scores from EmoTag1200."
  },
  {
   "aliases": [
    "BIG-bench empirical_judgments"
   ],
   "canonical_id": "empirical_judgments",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "empirical_judgments",
   "models_covered": 0,
   "name": "Empirical Judgments",
   "reasons": [],
   "summary": "A 99-item BIG-bench task that labels an English sentence as asserting a causal, correlative, or merely conceptual relation."
  },
  {
   "aliases": [],
   "canonical_id": "enem_challenge",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "enem_challenge",
   "models_covered": 0,
   "name": "ENEM Challenge",
   "reasons": [],
   "summary": "1,432 multiple-choice questions from Brazil's national secondary-school exam (ENEM), spanning 2009-2017 and 2022-2023, across humanities, languages, sciences and mathematics."
  },
  {
   "aliases": [
    "BIG-bench english_proverbs"
   ],
   "canonical_id": "english_proverbs",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "english_proverbs",
   "models_covered": 0,
   "name": "English Proverbs",
   "reasons": [],
   "summary": "A 34-item BIG-bench task that picks which English proverb best fits a short anecdote, with three to six choices."
  },
  {
   "aliases": [
    "english_russian_proverbs_json_multiple_choice",
    "BIG-bench english_russian_proverbs"
   ],
   "canonical_id": "english_russian_proverbs",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "english_russian_proverbs",
   "models_covered": 0,
   "name": "English to Russian Proverbs",
   "reasons": [],
   "summary": "An 80-item BIG-bench task that picks the Russian proverb closest in non-literal meaning to an English proverb."
  },
  {
   "aliases": [
    "BIG-bench entailed_polarity"
   ],
   "canonical_id": "entailed_polarity",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "entailed_polarity",
   "models_covered": 0,
   "name": "Entailed Polarity",
   "reasons": [],
   "summary": "A BIG-bench yes/no task that asks what polarity an English implicative verb entails for a simple follow-up question."
  },
  {
   "aliases": [
    "BIG-bench entailed_polarity_hindi"
   ],
   "canonical_id": "entailed_polarity_hindi",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "entailed_polarity_hindi",
   "models_covered": 0,
   "name": "Entailed Polarity in Hindi",
   "reasons": [],
   "summary": "A Hindi BIG-bench yes/no task that asks what polarity an implicative verb entails, translated from entailed_polarity."
  },
  {
   "aliases": [
    "entity_data_imputation",
    "HELM DataImputation",
    "DataImputation"
   ],
   "canonical_id": "entity_data_imputation",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "entity_data_imputation",
   "models_covered": 0,
   "name": "Data imputation (HELM)",
   "reasons": [],
   "summary": "HELM generation task that fills one missing column in a serialized product or restaurant row, scored with quasi-exact match."
  },
  {
   "aliases": [
    "entity_matching",
    "HELM EntityMatching",
    "EntityMatching"
   ],
   "canonical_id": "entity_matching",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "entity_matching",
   "models_covered": 0,
   "name": "Entity matching (HELM)",
   "reasons": [],
   "summary": "HELM generation task that asks whether two serialized table rows refer to the same entity, scored with exact match on Yes/No."
  },
  {
   "aliases": [
    "BIG-bench epistemic_reasoning"
   ],
   "canonical_id": "epistemic_reasoning",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "epistemic_reasoning",
   "models_covered": 0,
   "name": "Epistemic Reasoning",
   "reasons": [],
   "summary": "A 2,000-item BIG-bench NLI task that asks whether a nested knowledge or belief premise entails a hypothesis, scored as two-way multiple choice."
  },
  {
   "aliases": [
    "Emotional Intelligence Benchmark for Large Language Models"
   ],
   "canonical_id": "eq_bench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "eq_bench",
   "models_covered": 0,
   "name": "EQ-Bench",
   "reasons": [],
   "summary": "Rates the intensity of four emotions a character feels at the end of a GPT-4-written dialogue, scored by distance from an author-set reference; the version harnesses run today is now legacy on the publisher's own site."
  },
  {
   "aliases": [
    "EQ-bench_ca"
   ],
   "canonical_id": "eq_bench_ca",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "eq_bench_ca",
   "models_covered": 0,
   "name": "EQ-Bench (Catalan)",
   "reasons": [],
   "summary": "Catalan translation and cultural adaptation of EQ-Bench version 2's 167-171 question emotion-intensity rating task, released by the Barcelona Supercomputing Center."
  },
  {
   "aliases": [
    "EQ-bench_es"
   ],
   "canonical_id": "eq_bench_es",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "eq_bench_es",
   "models_covered": 0,
   "name": "EQ-Bench (Spanish)",
   "reasons": [],
   "summary": "Spanish translation and cultural adaptation of EQ-Bench version 2's 168-171 question emotion-intensity rating task, released by the Barcelona Supercomputing Center."
  },
  {
   "aliases": [],
   "canonical_id": "es_memeval",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "es_memeval",
   "models_covered": 0,
   "name": "ES-MemEval",
   "reasons": [],
   "summary": "ES-MemEval evaluates long-term conversational memory for personalized emotional support across extraction, temporal reasoning, conflict detection, abstention, and user modeling."
  },
  {
   "aliases": [
    "Spanish Bias Benchmark for Question Answering",
    "Spanish BBQ",
    "EsBBQ"
   ],
   "canonical_id": "esbbq",
   "category": "safety",
   "disposition": "unassessed",
   "id": "esbbq",
   "models_covered": 0,
   "name": "EsBBQ (Spanish Bias Benchmark for Question Answering)",
   "reasons": [],
   "summary": "Spanish multiple-choice QA, parallel with CaBBQ, testing stereotype use under ambiguous context and accuracy once the context names the answer."
  },
  {
   "aliases": [],
   "canonical_id": "eus_exams",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "eus_exams",
   "models_covered": 0,
   "name": "EusExams",
   "reasons": [],
   "summary": "35,727 multiple-choice questions from real Basque public-service admission exams, parallel in Basque (16,774) and Spanish (18,953), from the Latxa evaluation suite."
  },
  {
   "aliases": [
    "EusProf"
   ],
   "canonical_id": "eus_proficiency",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "eus_proficiency",
   "models_covered": 0,
   "name": "EusProficiency",
   "reasons": [],
   "summary": "5,169 four-option Basque questions from the EGA C1 qualifying paper (atarikoa, 1998-2008), one of four Latxa evaluation-suite tasks."
  },
  {
   "aliases": [],
   "canonical_id": "eus_reading",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "eus_reading",
   "models_covered": 0,
   "name": "EusReading",
   "reasons": [],
   "summary": "352 Basque reading-comprehension items from EGA irakurmena papers (1998-2008), with long passages that the Latxa paper used as a 1-shot long-context probe."
  },
  {
   "aliases": [
    "TriviaEus"
   ],
   "canonical_id": "eus_trivia",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "eus_trivia",
   "models_covered": 0,
   "name": "EusTrivia",
   "reasons": [],
   "summary": "1,715 Basque trivia questions (2-4 options, 3.84 on average) from online sources, 56.3% elementary, with a large Basque Country, language and culture share."
  },
  {
   "aliases": [
    "evalita-mp",
    "Evalita LLM"
   ],
   "canonical_id": "evalita_llm",
   "category": "composite",
   "disposition": "unassessed",
   "id": "evalita_llm",
   "models_covered": 0,
   "name": "Evalita-LLM",
   "reasons": [],
   "summary": "10 native-Italian NLP tasks, mostly drawn from the long-running Evalita campaign, each scored across six or two prompt variants to measure a model's sensitivity to prompt wording."
  },
  {
   "aliases": [
    "The Essential, the Excessive, and the Extraneous"
   ],
   "canonical_id": "evaluating_information_essentiality",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "evaluating_information_essentiality",
   "models_covered": 0,
   "name": "Evaluating Information Essentiality",
   "reasons": [],
   "summary": "A tiny, 68-item BIG-bench task modelled on GMAT-style data-sufficiency questions, testing whether a model can tell which of two statements are necessary, sufficient, or redundant to answer a question."
  },
  {
   "aliases": [],
   "canonical_id": "evo_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "evo_bench",
   "models_covered": 0,
   "name": "Evo-Bench",
   "reasons": [],
   "summary": "Evo-Bench evaluates whether language models can autonomously evolve agent harnesses across Search, Office and General domains."
  },
  {
   "aliases": [],
   "canonical_id": "evo_eval",
   "category": "coding",
   "disposition": "unassessed",
   "id": "evo_eval",
   "models_covered": 0,
   "name": "EvoEval",
   "reasons": [],
   "summary": "EvoEval evaluates code generation on evolved HumanEval-style problems across difficult, creative, subtle, combined, and tool-use domains."
  },
  {
   "aliases": [
    "Elements of World Knowledge",
    "EWoK-core-1.0",
    "ewok-core"
   ],
   "canonical_id": "ewok",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "ewok",
   "models_covered": 0,
   "name": "EWoK (Elements of World Knowledge)",
   "reasons": [],
   "summary": "EWoK-core-1.0 is a 4,374-item English set that tests whether a model matches a target sentence to the more plausible of two world-knowledge contexts."
  },
  {
   "aliases": [
    "EXAMS",
    "EXAMS-QA"
   ],
   "canonical_id": "exams_multilingual",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "exams_multilingual",
   "models_covered": 0,
   "name": "EXAMS (Multilingual)",
   "reasons": [],
   "summary": "24,143 real high-school exam questions across 16 languages and 24 subjects (Hardalov et al., 2020); arabic_exams documents the AceGPT-repackaged Arabic slice of this same corpus."
  },
  {
   "aliases": [
    "BIG-bench fact_checker",
    "Fact-Checking"
   ],
   "canonical_id": "fact_checker",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "fact_checker",
   "models_covered": 0,
   "name": "Fact-Checking (BIG-bench)",
   "reasons": [],
   "summary": "A BIG-bench true/false fact-checking task over Wikipedia FEVER claims, COVID-19 scientific claims, and Politifact items."
  },
  {
   "aliases": [
    "BIG-bench factuality_of_summary",
    "Factuality"
   ],
   "canonical_id": "factuality_of_summary",
   "category": "generation",
   "disposition": "unassessed",
   "id": "factuality_of_summary",
   "models_covered": 0,
   "name": "Factuality of Summary",
   "reasons": [],
   "summary": "A programmatic BIG-bench probe that picks the factual news summary among candidates using PMI of the summary given the document versus the summary alone."
  },
  {
   "aliases": [
    "FINE",
    "Fake alIgNment Evaluation",
    "fake_safety"
   ],
   "canonical_id": "fake_alignment",
   "category": "safety",
   "disposition": "unassessed",
   "id": "fake_alignment",
   "models_covered": 0,
   "name": "Fake Alignment (FINE)",
   "reasons": [],
   "summary": "FINE compares a model's open-ended safety answer with swapped multiple-choice safety options to score consistency (CS) and consistent safety (CSS)."
  },
  {
   "aliases": [
    "BIG-bench fantasy_reasoning"
   ],
   "canonical_id": "fantasy_reasoning",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "fantasy_reasoning",
   "models_covered": 0,
   "name": "Fantasy Reasoning",
   "reasons": [],
   "summary": "A 201-item BIG-bench Yes/No task that asks models to reason inside fantasy premises that violate real-world common sense."
  },
  {
   "aliases": [
    "BASED FDA",
    "EVAPORATE FDA"
   ],
   "canonical_id": "fda",
   "category": "domain",
   "disposition": "unassessed",
   "id": "fda",
   "models_covered": 0,
   "name": "FDA (BASED information extraction)",
   "reasons": [],
   "summary": "Zero-shot extraction of labelled values from chunked FDA 510(k) PDFs; lm-eval scores whether the gold value appears in a short continuation."
  },
  {
   "aliases": [
    "Language Generation from Structured Data and Schema Descriptions"
   ],
   "canonical_id": "few_shot_nlg",
   "category": "generation",
   "disposition": "unassessed",
   "id": "few_shot_nlg",
   "models_covered": 0,
   "name": "Few-shot NLG (BIG-bench)",
   "reasons": [],
   "summary": "A 153-example BIG-bench data-to-text task: verbalise unique dialogue-act and slot frames using schema descriptions, scored with BLEURT."
  },
  {
   "aliases": [
    "Few-shot CLUE",
    "Chinese Few-shot Learning Evaluation Benchmark"
   ],
   "canonical_id": "fewclue",
   "category": "composite",
   "disposition": "unassessed",
   "id": "fewclue",
   "models_covered": 0,
   "name": "FewCLUE (Chinese Few-shot Learning Evaluation Benchmark)",
   "reasons": [],
   "summary": "A Chinese few-shot NLU benchmark: nine tasks learned from 8-32 labelled examples per class across five parallel splits, so a score measures few-shot learning rather than full-data task competence."
  },
  {
   "aliases": [],
   "canonical_id": "fewclue_bustm",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "fewclue_bustm",
   "models_covered": 0,
   "name": "FewCLUE: BUSTM (Dialogue Short Text Matching)",
   "reasons": [],
   "summary": "FewCLUE's dialogue short-text matching task: judge whether two short colloquial Chinese sentences share the same intent, learned from 32 labelled training pairs."
  },
  {
   "aliases": [],
   "canonical_id": "fewclue_chid",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "fewclue_chid",
   "models_covered": 0,
   "name": "FewCLUE: CHID (Chinese Idiom Cloze Test)",
   "reasons": [],
   "summary": "FewCLUE's Chinese idiom cloze task: pick the idiom that fits a masked slot from seven near-synonym candidates, learned from 42 labelled training examples."
  },
  {
   "aliases": [],
   "canonical_id": "fewclue_cluewsc",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "fewclue_cluewsc",
   "models_covered": 0,
   "name": "FewCLUE: CLUEWSC (Winograd Schema Coreference)",
   "reasons": [],
   "summary": "FewCLUE's Winograd Schema task: judge whether a marked pronoun refers to a marked noun phrase in a Chinese sentence, learned from 32 labelled training examples."
  },
  {
   "aliases": [],
   "canonical_id": "fewclue_csl",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "fewclue_csl",
   "models_covered": 0,
   "name": "FewCLUE: CSL (Keyword Recognition)",
   "reasons": [],
   "summary": "FewCLUE's keyword-recognition task: judge whether every keyword listed for a Chinese academic abstract is genuine, learned from 32 labelled training examples."
  },
  {
   "aliases": [],
   "canonical_id": "fewclue_eprstmt",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "fewclue_eprstmt",
   "models_covered": 0,
   "name": "FewCLUE: EPRSTMT (E-commerce Sentiment Analysis)",
   "reasons": [],
   "summary": "FewCLUE's sentiment task: classify a Chinese e-commerce product review as positive or negative, learned from 32 labelled training reviews."
  },
  {
   "aliases": [],
   "canonical_id": "fewclue_ocnli_fc",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "fewclue_ocnli_fc",
   "models_covered": 0,
   "name": "FewCLUE: OCNLI-FC (Natural Language Inference)",
   "reasons": [],
   "summary": "FewCLUE's few-shot cut of OCNLI: classify a Chinese premise-hypothesis pair as entailment, neutral or contradiction, learned from 32 labelled training pairs."
  },
  {
   "aliases": [],
   "canonical_id": "fewclue_tnews",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "fewclue_tnews",
   "models_covered": 0,
   "name": "FewCLUE: TNEWS (Short News Classification)",
   "reasons": [],
   "summary": "FewCLUE's short news classification task: sort a Chinese Toutiao headline into one of 15 categories, learned from 240 labelled training examples."
  },
  {
   "aliases": [],
   "canonical_id": "figure_of_speech_detection",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "figure_of_speech_detection",
   "models_covered": 0,
   "name": "Figure of Speech Detection (BIG-bench)",
   "reasons": [],
   "summary": "A 59-item BIG-bench task that classifies an English sentence into one of ten figures of speech."
  },
  {
   "aliases": [],
   "canonical_id": "fin_qa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "fin_qa",
   "models_covered": 0,
   "name": "FinQA",
   "reasons": [],
   "summary": "8,281 expert-written QA pairs over S&P 500 earnings-report excerpts, scored by checking a generated arithmetic reasoning program rather than just a final number."
  },
  {
   "aliases": [],
   "canonical_id": "financebench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "financebench",
   "models_covered": 0,
   "name": "FinanceBench",
   "reasons": [],
   "summary": "Patronus AI's open-book financial QA test of 10,231 questions over 40 companies' public filings; only a 150-question human-graded sample is publicly released with answers."
  },
  {
   "aliases": [
    "FinanceIQ\uff08\u4e2d\u6587\u91d1\u878d\u9886\u57df\u77e5\u8bc6\u8bc4\u4f30\u6570\u636e\u96c6\uff09"
   ],
   "canonical_id": "financeiq",
   "category": "domain",
   "disposition": "unassessed",
   "id": "financeiq",
   "models_covered": 0,
   "name": "FinanceIQ",
   "reasons": [],
   "summary": "7,173 Chinese multiple-choice questions across 10 financial-licensing exam subjects, GPT-4-paraphrased and option-shuffled by its publisher specifically to resist pretraining leakage."
  },
  {
   "aliases": [
    "FinancialPhrasebank",
    "Financial Phrase Bank"
   ],
   "canonical_id": "financial_phrasebank",
   "category": "domain",
   "disposition": "unassessed",
   "id": "financial_phrasebank",
   "models_covered": 0,
   "name": "Financial PhraseBank",
   "reasons": [],
   "summary": "Aalto's three-class sentiment set of English financial-news sentences; HELM generates a label and reports weighted F1 on a 70/30 split."
  },
  {
   "aliases": [
    "FinBench (FinPT)"
   ],
   "canonical_id": "finbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "finbench",
   "models_covered": 13,
   "name": "FinBench",
   "reasons": [],
   "summary": "Ten Kaggle-sourced tabular datasets, turned into natural-language customer profiles, that test whether a model flags credit default, fraud or customer-churn risk."
  },
  {
   "aliases": [
    "Formal Logic Deduction",
    "FLD.v2",
    "FLD-star",
    "FLD\u2605"
   ],
   "canonical_id": "fld",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "fld",
   "models_covered": 0,
   "name": "FLD (Formal Logic Deduction)",
   "reasons": [],
   "summary": "Hitachi's synthetic deduction set: given invented facts and a hypothesis, choose proved, disproved, or unknown without using world knowledge."
  },
  {
   "aliases": [
    "FLORES",
    "Flores-200",
    "FLORES+",
    "No Language Left Behind evaluation benchmark"
   ],
   "canonical_id": "flores",
   "category": "translation",
   "disposition": "unassessed",
   "id": "flores",
   "models_covered": 0,
   "name": "FLORES-200",
   "reasons": [],
   "summary": "A human-translated, sentence-aligned evaluation set spanning 200 languages, used to score machine-translation quality between any language pair."
  },
  {
   "aliases": [
    "flores200 eng_Latn-deu_Latn"
   ],
   "canonical_id": "flores_en_de",
   "category": "translation",
   "disposition": "unassessed",
   "id": "flores_en_de",
   "models_covered": 15,
   "name": "FLORES English-to-German",
   "reasons": [],
   "summary": "BLEU score for English-to-German translation on the FLORES-200 devtest set, as reported in this repository's model cards."
  },
  {
   "aliases": [
    "flores200 eng_Latn-spa_Latn"
   ],
   "canonical_id": "flores_en_es",
   "category": "translation",
   "disposition": "unassessed",
   "id": "flores_en_es",
   "models_covered": 15,
   "name": "FLORES English-to-Spanish",
   "reasons": [],
   "summary": "BLEU score for English-to-Spanish translation on the FLORES-200 devtest set, as reported in this repository's model cards."
  },
  {
   "aliases": [
    "flores200 eng_Latn-jpn_Jpan"
   ],
   "canonical_id": "flores_en_ja",
   "category": "translation",
   "disposition": "unassessed",
   "id": "flores_en_ja",
   "models_covered": 15,
   "name": "FLORES English-to-Japanese",
   "reasons": [],
   "summary": "BLEU score for English-to-Japanese translation on the FLORES-200 devtest set, as reported in this repository's model cards."
  },
  {
   "aliases": [
    "flores200 eng_Latn-zho"
   ],
   "canonical_id": "flores_en_zh",
   "category": "translation",
   "disposition": "unassessed",
   "id": "flores_en_zh",
   "models_covered": 15,
   "name": "FLORES English-to-Chinese",
   "reasons": [],
   "summary": "BLEU score for English-to-Chinese translation on the FLORES-200 devtest set, as reported in this repository's model cards."
  },
  {
   "aliases": [],
   "canonical_id": "forecastbench_sim",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "forecastbench_sim",
   "models_covered": 0,
   "name": "ForecastBench-Sim",
   "reasons": [],
   "summary": "ForecastBench-Sim evaluates probabilistic forecasting on hidden future states in Freeciv game rollouts."
  },
  {
   "aliases": [],
   "canonical_id": "forecasting_subquestions",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "forecasting_subquestions",
   "models_covered": 0,
   "name": "Forecasting Subquestions (BIG-bench)",
   "reasons": [],
   "summary": "A programmatic BIG-bench task that scores log probability of human-written subquestions for unresolved Metaculus forecasting items."
  },
  {
   "aliases": [
    "formal_fallacies",
    "Formal Fallacies and Syllogisms with Negation"
   ],
   "canonical_id": "formal_fallacies_syllogisms_negation",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "formal_fallacies_syllogisms_negation",
   "models_covered": 0,
   "name": "Formal Fallacies Syllogisms Negation (BIG-bench)",
   "reasons": [],
   "summary": "14,200 synthetic arguments that ask whether a text is deductively valid or invalid, with a focus on negation; also a 250-item BIG-bench Hard slice."
  },
  {
   "aliases": [
    "Frontier Risk Evaluation for National Security and Public Safety",
    "ScaleAI/fortress_public"
   ],
   "canonical_id": "fortress",
   "category": "safety",
   "disposition": "unassessed",
   "id": "fortress",
   "models_covered": 0,
   "name": "FORTRESS",
   "reasons": [],
   "summary": "500 public NSPS adversarial prompts with instance rubrics, scored as average risk (ARS) plus over-refusal (ORS) on 500 paired benign prompts."
  },
  {
   "aliases": [
    "French Bench"
   ],
   "canonical_id": "french_bench",
   "category": "composite",
   "disposition": "unassessed",
   "id": "french_bench",
   "models_covered": 0,
   "name": "FrenchBench",
   "reasons": [],
   "summary": "An lm-evaluation-harness suite of about twenty French tasks mixing native resources (FQuAD, French Trivia, OrangeSum) with GPT-3.5-translated ones (HellaSwag, ARC-Challenge), from the CroissantLLM paper."
  },
  {
   "aliases": [
    "FrontierCS",
    "FrontierCS: Evolving Challenges for Evolving Intelligence"
   ],
   "canonical_id": "frontier_cs",
   "category": "coding",
   "disposition": "unassessed",
   "id": "frontier_cs",
   "models_covered": 0,
   "name": "Frontier-CS",
   "reasons": [],
   "summary": "Open-ended CS problems with continuous partial scoring on algorithmic and research tracks; Inspect pins 238 items while later releases add more."
  },
  {
   "aliases": [
    "openai/frontierscience"
   ],
   "canonical_id": "frontierscience",
   "category": "domain",
   "disposition": "unassessed",
   "id": "frontierscience",
   "models_covered": 0,
   "name": "FrontierScience",
   "reasons": [],
   "summary": "OpenAI expert-level science suite: 100 Olympiad short-answer items and 60 Research rubric tasks in physics, chemistry, and biology."
  },
  {
   "aliases": [
    "FrontierScience Research"
   ],
   "canonical_id": "frontierscience_research",
   "category": "domain",
   "disposition": "unassessed",
   "id": "frontierscience_research",
   "models_covered": 1,
   "name": "FrontierScience \u2014 Research track",
   "reasons": [],
   "summary": "60 open-ended, PhD-level scientific research subtasks in physics, chemistry and biology, graded on a 10-point rubric."
  },
  {
   "aliases": [],
   "canonical_id": "full_duplex_bench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "full_duplex_bench",
   "models_covered": 0,
   "name": "Full-Duplex-Bench",
   "reasons": [],
   "summary": "Full-Duplex-Bench evaluates spoken dialogue models on pause handling, backchanneling, turn-taking and interruption management."
  },
  {
   "aliases": [],
   "canonical_id": "full_duplex_bench_v3",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "full_duplex_bench_v3",
   "models_covered": 0,
   "name": "Full-Duplex-Bench-v3",
   "reasons": [],
   "summary": "Full-Duplex-Bench-v3 evaluates spoken language models under naturalistic human audio, disfluencies and chained tool use."
  },
  {
   "aliases": [
    "General AI Assistants",
    "GAIA: a benchmark for General AI Assistants",
    "gaia-benchmark/GAIA"
   ],
   "canonical_id": "gaia",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "gaia",
   "models_covered": 0,
   "name": "GAIA",
   "reasons": [],
   "summary": "466 real-world assistant questions needing tools and browsing; 300 test answers are held out and scoring is quasi-exact match."
  },
  {
   "aliases": [],
   "canonical_id": "galician_bench",
   "category": "composite",
   "disposition": "unassessed",
   "id": "galician_bench",
   "models_covered": 0,
   "name": "GalicianBench",
   "reasons": [],
   "summary": "The IberoBench suite for Galician: 11 lm-evaluation-harness task groups mixing three natively built Galician resources with eight translated from English or multilingual sources."
  },
  {
   "aliases": [
    "game-of-24",
    "ToT Game of 24",
    "Game24"
   ],
   "canonical_id": "game24",
   "category": "math",
   "disposition": "unassessed",
   "id": "game24",
   "models_covered": 0,
   "name": "Game of 24",
   "reasons": [],
   "summary": "Four-number puzzles that must equal 24; OpenCompass runs a five-puzzle Tree-of-Thoughts slice of the Princeton Game-of-24 set."
  },
  {
   "aliases": [
    "GaoKao MATH Answer Evaluation",
    "gaokao_math"
   ],
   "canonical_id": "gaokao_math",
   "category": "math",
   "disposition": "unassessed",
   "id": "gaokao_math",
   "models_covered": 0,
   "name": "GaoKaoMATH",
   "reasons": [],
   "summary": "OpenCompass LLM-judge pipeline that extracts and checks answers from Gaokao-style math writeups; the item file is not publicly hosted."
  },
  {
   "aliases": [
    "GaokaoBench",
    "Gaokao Bench",
    "Evaluating the Performance of Large Language Models on GAOKAO Benchmark"
   ],
   "canonical_id": "gaokaobench",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "gaokaobench",
   "models_covered": 0,
   "name": "GAOKAO-Bench",
   "reasons": [],
   "summary": "Real Chinese Gaokao exam questions from 2010-2022 (1,781 objective, 1,030 subjective), scored zero-shot and converted to the exam's own 750-point scale; most harnesses run only the objective subset."
  },
  {
   "aliases": [
    "GDM Dangerous Capabilities: Capture the Flag",
    "GDM Dangerous Capabilities: In House CTF"
   ],
   "canonical_id": "gdm_in_house_ctf",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "gdm_in_house_ctf",
   "models_covered": 0,
   "name": "GDM In-house CTF",
   "reasons": [],
   "summary": "Thirteen DeepMind in-house CTF tasks for a Kali bash agent; a challenge counts as solved if any of ten epochs captures the flag."
  },
  {
   "aliases": [
    "InterCode-CTF",
    "InterCode CTF",
    "inspect_evals/gdm_intercode_ctf"
   ],
   "canonical_id": "gdm_intercode_ctf",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "gdm_intercode_ctf",
   "models_covered": 0,
   "name": "GDM InterCode CTF",
   "reasons": [],
   "summary": "78 picoCTF tasks from InterCode-CTF: a tool-using agent must recover picoCTF{...} flags in Docker, as used in DeepMind's 2024 cyber evals."
  },
  {
   "aliases": [
    "GDM Dangerous Capabilities: Self-proliferation"
   ],
   "canonical_id": "gdm_self_proliferation",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "gdm_self_proliferation",
   "models_covered": 0,
   "name": "GDM Self-proliferation",
   "reasons": [],
   "summary": "Ten DeepMind agent tasks on email, cloud, wallets, and self-improvement, run with human approval in Inspect."
  },
  {
   "aliases": [
    "GDM Situational Awareness",
    "GDM Dangerous Capabilities: Self-reasoning"
   ],
   "canonical_id": "gdm_self_reasoning",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "gdm_self_reasoning",
   "models_covered": 0,
   "name": "GDM Self-reasoning",
   "reasons": [],
   "summary": "Eleven DeepMind challenges where an agent must notice and change its own config or tools to finish a job."
  },
  {
   "aliases": [
    "GDM Dangerous Capabilities: Stealth"
   ],
   "canonical_id": "gdm_stealth",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "gdm_stealth",
   "models_covered": 0,
   "name": "GDM Stealth",
   "reasons": [],
   "summary": "Google DeepMind stealth suite: tool-using agents try to hide policy-breaking actions from monitors."
  },
  {
   "aliases": [
    "GDP.pdf AA",
    "AA GDP.pdf",
    "GDP.pdf (Artificial Analysis)"
   ],
   "canonical_id": "gdp_pdf_aa",
   "category": "long-context",
   "disposition": "active",
   "id": "gdp_pdf_aa",
   "models_covered": 0,
   "name": "GDP.pdf-AA",
   "reasons": [
    "qualifying current coverage"
   ],
   "summary": "Artificial Analysis's independent GDP.pdf run: 100 professional PDF tasks, All-pass over 1,275 criteria, LiteParse text plus page images, GPT-5.6 Luna Medium judge."
  },
  {
   "aliases": [
    "GDP-val",
    "GDP val",
    "inspect_evals/gdpval"
   ],
   "canonical_id": "gdpval",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "gdpval",
   "models_covered": 0,
   "name": "GDPval",
   "reasons": [],
   "summary": "OpenAI's 220-task gold set of real occupation work products; models produce files that experts (or OpenAI's grader) compare with human deliverables."
  },
  {
   "aliases": [
    "GDPval-AA v2",
    "AA GDPval",
    "GDPval AA"
   ],
   "canonical_id": "gdpval_aa",
   "category": "agentic",
   "disposition": "active",
   "id": "gdpval_aa",
   "models_covered": 0,
   "name": "GDPval-AA",
   "reasons": [
    "qualifying current coverage"
   ],
   "summary": "Artificial Analysis independently scores OpenAI's 220-task GDPval gold set with pairwise Elo, shown as clamp((Elo-500)/2000)."
  },
  {
   "aliases": [
    "BIG-bench gem",
    "Various Generation Skills Benchmark"
   ],
   "canonical_id": "gem",
   "category": "generation",
   "disposition": "unassessed",
   "id": "gem",
   "models_covered": 0,
   "name": "GEM (BIG-bench)",
   "reasons": [],
   "summary": "A 14,802-example BIG-bench collection of 13 GEM generation subtasks, scored with ROUGE-Lsum, not the full GEM workshop suite."
  },
  {
   "aliases": [],
   "canonical_id": "genai_bench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "genai_bench",
   "models_covered": 0,
   "name": "GenAI-Bench",
   "reasons": [],
   "summary": "GenAI-Bench evaluates compositional text-to-image and text-to-video generation with human ratings and automated metric comparisons."
  },
  {
   "aliases": [
    "BIG-bench gender_inclusive_sentences_german",
    "Transforming German Sentences to Gender-Inclusive Forms"
   ],
   "canonical_id": "gender_inclusive_sentences_german",
   "category": "generation",
   "disposition": "unassessed",
   "id": "gender_inclusive_sentences_german",
   "models_covered": 0,
   "name": "Gender Inclusive Sentences German",
   "reasons": [],
   "summary": "200 German news sentences rewritten into gender-inclusive forms; BIG-bench scores exact string match against author targets."
  },
  {
   "aliases": [
    "gender sensitivity test Chinese"
   ],
   "canonical_id": "gender_sensitivity_chinese",
   "category": "safety",
   "disposition": "unassessed",
   "id": "gender_sensitivity_chinese",
   "models_covered": 0,
   "name": "Gender Sensitivity Test - Chinese",
   "reasons": [],
   "summary": "A Chinese BIG-bench task that scores occupation-title gender bias and accuracy at inferring gender from gendered terms and names."
  },
  {
   "aliases": [
    "gender sensitivity test English"
   ],
   "canonical_id": "gender_sensitivity_english",
   "category": "safety",
   "disposition": "unassessed",
   "id": "gender_sensitivity_english",
   "models_covered": 0,
   "name": "Gender Sensitivity Test - English",
   "reasons": [],
   "summary": "An English BIG-bench task that scores occupation-title gender bias, gender identification from names and terms, and PTB perplexity."
  },
  {
   "aliases": [],
   "canonical_id": "general365",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "general365",
   "models_covered": 0,
   "name": "General365",
   "reasons": [],
   "summary": "365 seed logic-and-reasoning problems expanded to 1,095 variants across eight categories, designed to test reasoning that needs only K-12 knowledge; even the top model reached 62.8% at release."
  },
  {
   "aliases": [
    "BIG-bench general_knowledge"
   ],
   "canonical_id": "general_knowledge",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "general_knowledge",
   "models_covered": 0,
   "name": "General Knowledge",
   "reasons": [],
   "summary": "A 70-item BIG-bench multiple-choice set of child-level and oddball English facts, scored as multiple_choice_grade."
  },
  {
   "aliases": [
    "BIG-bench geometric_shapes",
    "BBH geometric_shapes"
   ],
   "canonical_id": "geometric_shapes",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "geometric_shapes",
   "models_covered": 0,
   "name": "Geometric Shapes",
   "reasons": [],
   "summary": "360 SVG path strings labelled with a shape name; BIG-bench scores multiple_choice_grade, and BBH keeps a 250-item slice."
  },
  {
   "aliases": [
    "Global MMLU",
    "CohereLabs/Global-MMLU",
    "CohereForAI/Global-MMLU"
   ],
   "canonical_id": "global_mmlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "global_mmlu",
   "models_covered": 0,
   "name": "Global-MMLU",
   "reasons": [],
   "summary": "MMLU's questions in 42 languages, with CS/CA labels on 2,850 items per language; lm-eval ships Lite (15 languages) and full groups."
  },
  {
   "aliases": [
    "General Language Understanding Evaluation benchmark"
   ],
   "canonical_id": "glue",
   "category": "composite",
   "disposition": "unassessed",
   "id": "glue",
   "models_covered": 0,
   "name": "GLUE (General Language Understanding Evaluation benchmark)",
   "reasons": [],
   "summary": "A nine-task English sentence-understanding suite that defined pre-LLM benchmarking from 2018; models exceeded its human baseline within about 14 months, and its own successor SuperGLUE replaced it."
  },
  {
   "aliases": [],
   "canonical_id": "glue_cola",
   "category": "composite",
   "disposition": "unassessed",
   "id": "glue_cola",
   "models_covered": 0,
   "name": "GLUE: CoLA (Corpus of Linguistic Acceptability)",
   "reasons": [],
   "summary": "GLUE's single-sentence task: judge whether an English sentence is grammatically acceptable, scored by Matthews correlation because the acceptable/unacceptable classes are unbalanced."
  },
  {
   "aliases": [],
   "canonical_id": "glue_mrpc",
   "category": "composite",
   "disposition": "unassessed",
   "id": "glue_mrpc",
   "models_covered": 0,
   "name": "GLUE: MRPC (Microsoft Research Paraphrase Corpus)",
   "reasons": [],
   "summary": "GLUE's paraphrase task: judge whether two English news sentences mean the same thing, scored by the mean of accuracy and F1 because the classes are unbalanced."
  },
  {
   "aliases": [],
   "canonical_id": "glue_qqp",
   "category": "composite",
   "disposition": "unassessed",
   "id": "glue_qqp",
   "models_covered": 0,
   "name": "GLUE: QQP (Quora Question Pairs)",
   "reasons": [],
   "summary": "GLUE's largest task: judge whether two Quora questions ask the same thing, scored by the mean of accuracy and F1 because the classes are unbalanced."
  },
  {
   "aliases": [
    "goal_step_wikihow",
    "WikiHow goal-step",
    "goal-step inference"
   ],
   "canonical_id": "goal_step_wikihow",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "goal_step_wikihow",
   "models_covered": 0,
   "name": "Goal-step WikiHow (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench JSON task with 7,053 wikiHow multiple-choice items on goal-step inference and step order, scored by multiple_choice_grade."
  },
  {
   "aliases": [
    "gold_commodity_news",
    "Gold Commodity News and Dimensions",
    "gold-commodity-news-and-dimensions"
   ],
   "canonical_id": "gold_commodity_news",
   "category": "domain",
   "disposition": "unassessed",
   "id": "gold_commodity_news",
   "models_covered": 0,
   "name": "Gold Commodity News (HELM)",
   "reasons": [],
   "summary": "HELM Enterprise yes/no classification of 11,412 gold-commodity headlines across nine information dimensions, scored by weighted F1."
  },
  {
   "aliases": [
    "GovRepcrs",
    "govrepcrs_gen",
    "GovReport CRS"
   ],
   "canonical_id": "govrepcrs",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "govrepcrs",
   "models_covered": 0,
   "name": "GovRepcrs (OpenCompass / GovReport CRS)",
   "reasons": [],
   "summary": "OpenCompass English summarization of Congressional Research Service reports from GovReport, scored with BLEU rather than ROUGE."
  },
  {
   "aliases": [
    "Graduate-Level Google-Proof Q&A Benchmark"
   ],
   "canonical_id": "gpqa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "gpqa",
   "models_covered": 0,
   "name": "GPQA",
   "reasons": [],
   "summary": "Graduate-level multiple-choice science questions in biology, physics and chemistry, built to resist internet lookup; the family behind the widely-reported GPQA Diamond subset."
  },
  {
   "aliases": [
    "GPQA-Diamond"
   ],
   "canonical_id": "gpqa_diamond",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "gpqa_diamond",
   "models_covered": 356,
   "name": "GPQA Diamond",
   "reasons": [],
   "summary": "198 PhD-written multiple-choice science questions built to resist lookup, the hardest subset of GPQA."
  },
  {
   "aliases": [
    "Best ChatGPT Prompts",
    "HELM grammar",
    "grammar:path",
    "best_chatgpt_prompts"
   ],
   "canonical_id": "grammar",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "grammar",
   "models_covered": 0,
   "name": "Grammar / Best ChatGPT Prompts (HELM Instruct)",
   "reasons": [],
   "summary": "HELM Instruct scenario that expands a CFG of Gridfiti ChatGPT prompts and scores free-form replies with a 1-5 Helpfulness critique."
  },
  {
   "aliases": [
    "Graphwalks"
   ],
   "canonical_id": "graphwalks",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "graphwalks",
   "models_covered": 0,
   "name": "GraphWalks",
   "reasons": [],
   "summary": "OpenAI's long-context eval that hides a directed graph of hashed node names in the prompt and asks the model to run a breadth-first search or list a node's parents."
  },
  {
   "aliases": [
    "GraphWalks BFS, 256K subset of 1M"
   ],
   "canonical_id": "graphwalks_bfs_256k_1m",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "graphwalks_bfs_256k_1m",
   "models_covered": 3,
   "name": "GraphWalks BFS (256K-1M context)",
   "reasons": [],
   "summary": "GraphWalks' breadth-first-search task, scored only on prompts from the dataset's longest file, spanning roughly 256K to 1M tokens of context."
  },
  {
   "aliases": [
    "GraphWalks Parents, 256K subset of 1M"
   ],
   "canonical_id": "graphwalks_parents_256k_1m",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "graphwalks_parents_256k_1m",
   "models_covered": 1,
   "name": "GraphWalks Parents (256K-1M context)",
   "reasons": [],
   "summary": "GraphWalks' parent-finding task, scored only on prompts from the dataset's longest file, spanning roughly 256K to 1M tokens of context."
  },
  {
   "aliases": [
    "gre_reading_comprehension",
    "BIG-bench GRE RC"
   ],
   "canonical_id": "gre_reading_comprehension",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "gre_reading_comprehension",
   "models_covered": 0,
   "name": "GRE Reading Comprehension (BIG-bench)",
   "reasons": [],
   "summary": "A 32-item BIG-bench GRE reading-comprehension task: English passages with 3- or 5-way questions, some with two gold answers."
  },
  {
   "aliases": [
    "Greek MMLU",
    "GMMLU"
   ],
   "canonical_id": "greekmmlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "greekmmlu",
   "models_covered": 0,
   "name": "GreekMMLU",
   "reasons": [],
   "summary": "21,805 multiple-choice questions natively sourced or authored in Greek from real exams across 45 subjects, built specifically to avoid the translation artefacts of machine-translated Greek benchmarks."
  },
  {
   "aliases": [
    "GroundCocoa",
    "ground_cocoa",
    "harsh147/GroundCocoa"
   ],
   "canonical_id": "groundcocoa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "groundcocoa",
   "models_covered": 0,
   "name": "GroundCocoa",
   "reasons": [],
   "summary": "Five-way multiple-choice flight-booking task that tests compositional and conditional reasoning over user constraints; 4,849 public test items."
  },
  {
   "aliases": [
    "Grade School Math 8K"
   ],
   "canonical_id": "gsm8k",
   "category": "math",
   "disposition": "unassessed",
   "id": "gsm8k",
   "models_covered": 130,
   "name": "GSM8K",
   "reasons": [],
   "summary": "8.5K grade-school math word problems scored by exact match on the final numeric answer, testing multi-step arithmetic reasoning."
  },
  {
   "aliases": [
    "gsm8k_contamination_ppl",
    "gsm8k-train-ppl",
    "gsm8k-test-ppl",
    "gsm8k-ref-ppl",
    "mock_gsm8k_test"
   ],
   "canonical_id": "gsm8k_contamination",
   "category": "math",
   "disposition": "unassessed",
   "id": "gsm8k_contamination",
   "models_covered": 0,
   "name": "GSM8K contamination (OpenCompass PPL probe)",
   "reasons": [],
   "summary": "OpenCompass perplexity probe comparing GSM8K train, GSM8K test, and a Skywork mock set to flag training-set overlap, not math accuracy."
  },
  {
   "aliases": [
    "gsm-hard",
    "GSM Hard",
    "gsmhard"
   ],
   "canonical_id": "gsm_hard",
   "category": "math",
   "disposition": "unassessed",
   "id": "gsm_hard",
   "models_covered": 0,
   "name": "GSM-Hard",
   "reasons": [],
   "summary": "GSM8K test problems with one number replaced by a large integer, scored by exact match on the recalculated numeric answer."
  },
  {
   "aliases": [],
   "canonical_id": "gui_cc",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "gui_cc",
   "models_covered": 0,
   "name": "GUI-CC",
   "reasons": [],
   "summary": "GUI-CC evaluates whether GUI world models preserve context across repeated interaction instead of only predicting plausible next screens."
  },
  {
   "aliases": [
    "HAE-RAE Bench",
    "HAE_RAE_BENCH",
    "HAERAE-BENCH",
    "HRB"
   ],
   "canonical_id": "haerae",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "haerae",
   "models_covered": 0,
   "name": "HAE-RAE Bench (lm-eval haerae)",
   "reasons": [],
   "summary": "lm-eval group for HAE-RAE Bench: five Korean multiple-choice tasks of native vocabulary and culture; the paper's reading-comprehension slice is omitted."
  },
  {
   "aliases": [],
   "canonical_id": "harm_bench",
   "category": "safety",
   "disposition": "unassessed",
   "id": "harm_bench",
   "models_covered": 0,
   "name": "HarmBench",
   "reasons": [],
   "summary": "A standardized red-teaming framework of 510 curated harmful behaviors testing whether attacks make a model comply; the metric is Attack Success Rate, so a lower score is the safety-desirable outcome."
  },
  {
   "aliases": [
    "HarmBenchGCGTransfer",
    "GCG-T",
    "HarmBench GCG-T"
   ],
   "canonical_id": "harm_bench_gcg_transfer",
   "category": "safety",
   "disposition": "unassessed",
   "id": "harm_bench_gcg_transfer",
   "models_covered": 0,
   "name": "HarmBench GCG-Transfer (GCG-T)",
   "reasons": [],
   "summary": "HarmBench behaviors attacked with GCG suffixes optimized once against four open models and transferred unchanged to the target; a lower Attack Success Rate is the safety-desirable outcome."
  },
  {
   "aliases": [
    "HeadQA"
   ],
   "canonical_id": "headqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "headqa",
   "models_covered": 0,
   "name": "HEAD-QA",
   "reasons": [],
   "summary": "HEAD-QA scores multiple-choice questions from real Spanish healthcare civil-service exams, released in matched Spanish and English versions."
  },
  {
   "aliases": [],
   "canonical_id": "healthbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "healthbench",
   "models_covered": 0,
   "name": "HealthBench",
   "reasons": [],
   "summary": "HealthBench grades a model's responses in realistic, multi-turn health conversations against physician-written rubrics, using another model as the judge."
  },
  {
   "aliases": [],
   "canonical_id": "healthbench_hard",
   "category": "domain",
   "disposition": "unassessed",
   "id": "healthbench_hard",
   "models_covered": 1,
   "name": "HealthBench Hard",
   "reasons": [],
   "summary": "The 1,000 hardest conversations from OpenAI's HealthBench, graded against physician-written rubrics of what a good health-related response should do."
  },
  {
   "aliases": [
    "HealthQA-BR",
    "healthqa-br",
    "Larxel/healthqa-br"
   ],
   "canonical_id": "healthqa_br",
   "category": "domain",
   "disposition": "unassessed",
   "id": "healthqa_br",
   "models_covered": 0,
   "name": "HealthQA-BR (HELM)",
   "reasons": [],
   "summary": "HELM wrap of HealthQA-BR: 5,632 Portuguese multiple-choice items from Brazilian medical licensing and residency exams."
  },
  {
   "aliases": [],
   "canonical_id": "heart",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "heart",
   "models_covered": 0,
   "name": "HEART",
   "reasons": [],
   "summary": "HEART compares human and LLM responses on the same multi-turn emotional-support conversations using blinded ratings and five interpersonal dimensions."
  },
  {
   "aliases": [],
   "canonical_id": "heart_bench",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "heart_bench",
   "models_covered": 0,
   "name": "HEART-Bench",
   "reasons": [],
   "summary": "HEART-Bench evaluates whether LLM agents preserve human-like personality and memory-consistent decisions across structured psychological scenarios."
  },
  {
   "aliases": [],
   "canonical_id": "hellaswag",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "hellaswag",
   "models_covered": 50,
   "name": "HellaSwag",
   "reasons": [],
   "summary": "Four-way multiple-choice test of commonsense sentence continuation, built by adversarial filtering to defeat models of its era."
  },
  {
   "aliases": [
    "HELM-Safety"
   ],
   "canonical_id": "helm_safety",
   "category": "safety",
   "disposition": "unassessed",
   "id": "helm_safety",
   "models_covered": 71,
   "name": "HELM Safety",
   "reasons": [],
   "summary": "Stanford CRFM leaderboard that scores a model by averaging its results across five existing safety benchmarks into one 0-1 number."
  },
  {
   "aliases": [
    "ETHICS",
    "Hendrycks ETHICS",
    "Aligning AI With Shared Human Values"
   ],
   "canonical_id": "hendrycks_ethics",
   "category": "safety",
   "disposition": "unassessed",
   "id": "hendrycks_ethics",
   "models_covered": 0,
   "name": "ETHICS (lm-eval hendrycks_ethics)",
   "reasons": [],
   "summary": "lm-eval tag/group for Hendrycks ETHICS: five English tasks of everyday moral judgment, scored as per-item accuracy on the public test split."
  },
  {
   "aliases": [
    "Helpful, Honest, & Harmless",
    "BIG-bench hhh_alignment",
    "hhh"
   ],
   "canonical_id": "hhh_alignment",
   "category": "safety",
   "disposition": "unassessed",
   "id": "hhh_alignment",
   "models_covered": 0,
   "name": "HHH Alignment (BIG-bench)",
   "reasons": [],
   "summary": "Anthropic BIG-bench task: pick the more helpful, honest, or harmless assistant reply in 221 binary comparisons."
  },
  {
   "aliases": [
    "High Low Guessing Game",
    "BIG-bench high_low_game"
   ],
   "canonical_id": "high_low_game",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "high_low_game",
   "models_covered": 0,
   "name": "High Low Game (BIG-bench)",
   "reasons": [],
   "summary": "Programmatic BIG-bench game: guess Alice's integer in [0, 100] from too-high / too-low feedback, scored by negated mean minimum error."
  },
  {
   "aliases": [
    "BIG-bench hindi_question_answering",
    "Hindi QnA"
   ],
   "canonical_id": "hindi_question_answering",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "hindi_question_answering",
   "models_covered": 0,
   "name": "Hindi Question Answering (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench Hindi span QA: 6,610 passage-question pairs from MLQA and XQuAD, scored with BLEU (also ROUGE and exact match)."
  },
  {
   "aliases": [
    "Hindu Mythology Trivia",
    "BIG-bench hindu_knowledge"
   ],
   "canonical_id": "hindu_knowledge",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "hindu_knowledge",
   "models_covered": 0,
   "name": "Hindu Knowledge (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench Lite memorization quiz: 175 multiple-choice questions on Hindu mythology, written from the authors' memory."
  },
  {
   "aliases": [
    "BIG-bench hinglish_toxicity"
   ],
   "canonical_id": "hinglish_toxicity",
   "category": "safety",
   "disposition": "unassessed",
   "id": "hinglish_toxicity",
   "models_covered": 0,
   "name": "Hinglish Toxicity (BIG-bench)",
   "reasons": [],
   "summary": "A 200-item BIG-bench yes/no task: judge whether a Hinglish (Hindi-English code-mixed) paragraph is toxic."
  },
  {
   "aliases": [],
   "canonical_id": "histoires_morales",
   "category": "safety",
   "disposition": "unassessed",
   "id": "histoires_morales",
   "models_covered": 0,
   "name": "Histoires Morales",
   "reasons": [],
   "summary": "Histoires Morales is a French moral-alignment dataset of normative and norm-divergent stories."
  },
  {
   "aliases": [
    "HLE"
   ],
   "canonical_id": "hle",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "hle",
   "models_covered": 6,
   "name": "Humanity's Last Exam",
   "reasons": [],
   "summary": "2,500 expert-written, closed-ended questions spanning dozens of academic subjects, built by CAIS and Scale AI to replace saturated benchmarks like MMLU."
  },
  {
   "aliases": [
    "HLE with search",
    "HLE tool-use"
   ],
   "canonical_id": "hle_tools",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "hle_tools",
   "models_covered": 5,
   "name": "Humanity's Last Exam (with tools)",
   "reasons": [],
   "summary": "The same 2,500 HLE questions scored when a model can search, fetch web pages and run code, instead of answering from its own knowledge alone."
  },
  {
   "aliases": [],
   "canonical_id": "hmmt2026",
   "category": "math",
   "disposition": "unassessed",
   "id": "hmmt2026",
   "models_covered": 0,
   "name": "HMMT 2026",
   "reasons": [],
   "summary": "HMMT 2026 is an OpenCompass mathematics evaluation using problems from the 2026 Harvard-MIT Mathematics Tournament."
  },
  {
   "aliases": [],
   "canonical_id": "hrm8k",
   "category": "math",
   "disposition": "unassessed",
   "id": "hrm8k",
   "models_covered": 0,
   "name": "HRM8K",
   "reasons": [],
   "summary": "HRM8K evaluates Korean and English mathematical reasoning on 8,011 parallel bilingual problems."
  },
  {
   "aliases": [
    "BIG-bench human_organs_senses"
   ],
   "canonical_id": "human_organs_senses",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "human_organs_senses",
   "models_covered": 0,
   "name": "Human Organs and Senses (BIG-bench)",
   "reasons": [],
   "summary": "A 42-item BIG-bench multiple-choice quiz on which human organ or sense a function belongs to."
  },
  {
   "aliases": [
    "OpenAI HumanEval"
   ],
   "canonical_id": "humaneval",
   "category": "coding",
   "disposition": "unassessed",
   "id": "humaneval",
   "models_covered": 200,
   "name": "HumanEval",
   "reasons": [],
   "summary": "Tests whether a model can write a correct Python function body from its signature and docstring, checked by executing hidden unit tests."
  },
  {
   "aliases": [
    "openai_humaneval_cn"
   ],
   "canonical_id": "humaneval_cn",
   "category": "coding",
   "disposition": "unassessed",
   "id": "humaneval_cn",
   "models_covered": 0,
   "name": "HumanEval-CN",
   "reasons": [],
   "summary": "OpenCompass's Chinese-instruction variant of HumanEval: the same 164 Python problems, evaluated with the task instruction given in Chinese rather than English."
  },
  {
   "aliases": [],
   "canonical_id": "humaneval_infilling",
   "category": "coding",
   "disposition": "unassessed",
   "id": "humaneval_infilling",
   "models_covered": 0,
   "name": "HumanEval-Infilling",
   "reasons": [],
   "summary": "Four fill-in-the-middle tasks built by masking spans of HumanEval's solutions; two (single/multi-line) come from the InCoder paper, two (random-span) were added by OpenAI's FIM paper."
  },
  {
   "aliases": [
    "HumanEval Plus"
   ],
   "canonical_id": "humaneval_plus",
   "category": "coding",
   "disposition": "unassessed",
   "id": "humaneval_plus",
   "models_covered": 0,
   "name": "HumanEval+",
   "reasons": [],
   "summary": "EvalPlus's stricter HumanEval, testing the same 164 Python problems against roughly 80x more unit tests so incorrect completions that pass the original suite get caught."
  },
  {
   "aliases": [],
   "canonical_id": "humaneval_pro",
   "category": "coding",
   "disposition": "unassessed",
   "id": "humaneval_pro",
   "models_covered": 0,
   "name": "HumanEval Pro",
   "reasons": [],
   "summary": "A harder successor to HumanEval that pairs each of its 164 problems with a second, more complex problem the model must solve by correctly invoking its own solution to the first."
  },
  {
   "aliases": [],
   "canonical_id": "humanevalx",
   "category": "coding",
   "disposition": "unassessed",
   "id": "humanevalx",
   "models_covered": 0,
   "name": "HumanEval-X",
   "reasons": [],
   "summary": "CodeGeeX's multilingual HumanEval: 820 hand-crafted problems across Python, C++, Java, JavaScript and Go, extending the same 164 tasks by hand rather than by machine translation."
  },
  {
   "aliases": [
    "Hungarian Math Exam",
    "HungarianExamMath",
    "Testing Language Models on a Held-Out High School National Finals Exam"
   ],
   "canonical_id": "hungarian_exam",
   "category": "math",
   "disposition": "unassessed",
   "id": "hungarian_exam",
   "models_covered": 0,
   "name": "Hungarian National HS Finals Exam (Mathematics)",
   "reasons": [],
   "summary": "Keiran Paster's late-2023 test of language models against that year's Hungarian national high-school mathematics finals, evaluated by hand because the exam was published after every tested model's training cutoff."
  },
  {
   "aliases": [
    "BBH hyperbaton",
    "BIG-bench hyperbaton"
   ],
   "canonical_id": "hyperbaton",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "hyperbaton",
   "models_covered": 0,
   "name": "Hyperbaton (BIG-bench)",
   "reasons": [],
   "summary": "A 50,000-item BIG-bench two-way quiz: pick which of two candidate sentences orders English adjectives correctly."
  },
  {
   "aliases": [
    "ICE"
   ],
   "canonical_id": "ice",
   "category": "generation",
   "disposition": "unassessed",
   "id": "ice",
   "models_covered": 0,
   "name": "International Corpus of English",
   "reasons": [],
   "summary": "ICE evaluates per-text perplexity across regional English corpora covering spoken and written language."
  },
  {
   "aliases": [
    "IWG",
    "icelandic-winogrande",
    "mideind/icelandic-winogrande"
   ],
   "canonical_id": "icelandic_winogrande",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "icelandic_winogrande",
   "models_covered": 0,
   "name": "Icelandic WinoGrande",
   "reasons": [],
   "summary": "Manually localized Icelandic WinoGrande schemas scored as two-way fill-in-the-blank accuracy; lm-eval runs the public train split."
  },
  {
   "aliases": [
    "BIG-bench identify_math_theorems"
   ],
   "canonical_id": "identify_math_theorems",
   "category": "math",
   "disposition": "unassessed",
   "id": "identify_math_theorems",
   "models_covered": 0,
   "name": "Identify Math Theorems (BIG-bench)",
   "reasons": [],
   "summary": "A 54-item BIG-bench task: read a purported LaTeX theorem and pick a correct statement, or the original if it is already true."
  },
  {
   "aliases": [
    "BIG-bench identify_odd_metaphor"
   ],
   "canonical_id": "identify_odd_metaphor",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "identify_odd_metaphor",
   "models_covered": 0,
   "name": "Identify Odd Metaphor (BIG-bench)",
   "reasons": [],
   "summary": "A 47-item BIG-bench task: among metaphorical sentences about one topic, pick the metaphor that would not also fit a second topic."
  },
  {
   "aliases": [],
   "canonical_id": "ifbench",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "ifbench",
   "models_covered": 0,
   "name": "IFBench",
   "reasons": [],
   "summary": "IFBench tests whether a model can follow verifiable output constraints it was not trained on, rather than the small fixed set most instruction-following benchmarks reuse."
  },
  {
   "aliases": [
    "Instruction-Following Eval",
    "Instruction-Following Evaluation for Large Language Models"
   ],
   "canonical_id": "ifeval",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "ifeval",
   "models_covered": 354,
   "name": "IFEval",
   "reasons": [],
   "summary": "IFEval scores whether a model's response obeys objectively checkable instructions, such as word counts or keyword frequency, using code rather than human or LLM judgment."
  },
  {
   "aliases": [
    "IFEval Catalan",
    "Instruction-Following Eval - Catalan"
   ],
   "canonical_id": "ifeval_ca",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "ifeval_ca",
   "models_covered": 0,
   "name": "IFEval_ca (Catalan IFEval)",
   "reasons": [],
   "summary": "A 541-prompt professional Catalan translation of IFEval, checked by a separately reimplemented Catalan instruction-verification codebase rather than IFEval's English checkers."
  },
  {
   "aliases": [
    "IFEval Spanish",
    "Instruction-Following Eval - Spanish"
   ],
   "canonical_id": "ifeval_es",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "ifeval_es",
   "models_covered": 0,
   "name": "IFEval_es (Spanish IFEval)",
   "reasons": [],
   "summary": "A 541-prompt professional Spanish translation of IFEval, checked by a separately reimplemented Spanish instruction-verification codebase rather than IFEval's English checkers."
  },
  {
   "aliases": [],
   "canonical_id": "ifevalcode",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "ifevalcode",
   "models_covered": 0,
   "name": "IFEvalCode",
   "reasons": [],
   "summary": "IFEvalCode scores code models on two independent checks per problem: functional correctness, and whether the code also obeys an explicit style or structural instruction, across eight languages."
  },
  {
   "aliases": [
    "IMDB",
    "Large Movie Review Dataset",
    "aclImdb"
   ],
   "canonical_id": "imdb",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "imdb",
   "models_covered": 0,
   "name": "IMDb (Large Movie Review Dataset)",
   "reasons": [],
   "summary": "50,000 polarised IMDb movie reviews for binary sentiment classification, from a 2011 paper; frontier models sit far past the ceiling, so it now serves mainly as a robustness and calibration check."
  },
  {
   "aliases": [
    "imdb_pt",
    "maritaca-ai/imdb_pt",
    "IMDB PT-BR"
   ],
   "canonical_id": "imdb_ptbr",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "imdb_ptbr",
   "models_covered": 0,
   "name": "IMDb PT-BR (HELM)",
   "reasons": [],
   "summary": "HELM scenario that classifies Maritaca's Portuguese IMDb reviews as positivo or negativo; the scored test file has 5,000 balanced rows."
  },
  {
   "aliases": [
    "BIG-bench implicatures"
   ],
   "canonical_id": "implicatures",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "implicatures",
   "models_covered": 0,
   "name": "Implicatures (BIG-bench)",
   "reasons": [],
   "summary": "A 492-item BIG-bench task: decide whether Speaker 2's reply to a yes/no question means yes or no."
  },
  {
   "aliases": [
    "BIG-bench implicit_relations",
    "Implicit Interpersonal Relations"
   ],
   "canonical_id": "implicit_relations",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "implicit_relations",
   "models_covered": 0,
   "name": "Implicit Relations (BIG-bench)",
   "reasons": [],
   "summary": "An 85-item BIG-bench task: read a short English passage and pick the relation of X to Y from 25 labels."
  },
  {
   "aliases": [
    "BIG-bench indic_cause_and_effect"
   ],
   "canonical_id": "indic_cause_and_effect",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "indic_cause_and_effect",
   "models_covered": 0,
   "name": "Indic Cause and Effect (BIG-bench)",
   "reasons": [],
   "summary": "A 585-query BIG-bench task: pick causal direction in Bengali, Hindi, or Malayalam under three prompt formats."
  },
  {
   "aliases": [
    "INDIC DIALECT",
    "INDIC-DIALECT",
    "Indic Dialect"
   ],
   "canonical_id": "indic_dialect",
   "category": "composite",
   "disposition": "unassessed",
   "id": "indic_dialect",
   "models_covered": 0,
   "name": "INDIC-DIALECT",
   "reasons": [],
   "summary": "Multi-task Hindi and Odia dialect suite: 11-way classification, translation MCQ, and dialect-standard MT over 13,000 native-speaker sentence pairs."
  },
  {
   "aliases": [
    "IndicDiarBench",
    "sarvamai/indic-diarbench"
   ],
   "canonical_id": "indic_diarbench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "indic_diarbench",
   "models_covered": 0,
   "name": "Indic DiarBench",
   "reasons": [],
   "summary": "Joint diarization and speaker-attributed ASR across all 22 scheduled Indian languages on about 108 hours of overlapping multi-speaker audio."
  },
  {
   "aliases": [
    "INDICXNLI",
    "Divyanshu/indicxnli"
   ],
   "canonical_id": "indicxnli",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "indicxnli",
   "models_covered": 0,
   "name": "IndicXNLI",
   "reasons": [],
   "summary": "Machine-translated XNLI for 11 Indic languages; lm-evaluation-harness currently scores only Gujarati three-way NLI on Divyanshu/indicxnli."
  },
  {
   "aliases": [
    "inference-ppl",
    "OpenCompass InferencePPL"
   ],
   "canonical_id": "inference_ppl",
   "category": "generation",
   "disposition": "unassessed",
   "id": "inference_ppl",
   "models_covered": 0,
   "name": "Inference-PPL",
   "reasons": [],
   "summary": "OpenCompass metric that averages token NLL only on labeled positions, shipped with a local cn-reasoning-val example rather than a fixed public test."
  },
  {
   "aliases": [],
   "canonical_id": "infinite_bench_en_mc",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "infinite_bench_en_mc",
   "models_covered": 0,
   "name": "\u221eBench: En.MC (English Multiple-Choice)",
   "reasons": [],
   "summary": "\u221eBench's English multiple-choice split: pick the correct answer among four options after reading a novel averaging around 184K tokens, testing aggregation and filtering, not just retrieval."
  },
  {
   "aliases": [],
   "canonical_id": "infinite_bench_en_qa",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "infinite_bench_en_qa",
   "models_covered": 0,
   "name": "\u221eBench: En.QA (English Question Answering)",
   "reasons": [],
   "summary": "\u221eBench's open-ended English QA split: answer a free-text question after reading a novel averaging around 193K tokens, scored by token-level F1 against the reference answer."
  },
  {
   "aliases": [],
   "canonical_id": "infinite_bench_en_sum",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "infinite_bench_en_sum",
   "models_covered": 0,
   "name": "\u221eBench: En.Sum (English Summarisation)",
   "reasons": [],
   "summary": "\u221eBench's summarisation split: produce a concise summary of a full English novel averaging around 104K tokens, scored by ROUGE-L-Sum against a web-sourced reference summary."
  },
  {
   "aliases": [
    "InfiniteBench",
    "\u221eBench"
   ],
   "canonical_id": "infinitebench",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "infinitebench",
   "models_covered": 0,
   "name": "\u221eBench (InfiniteBench)",
   "reasons": [],
   "summary": "\u221eBench tests long-context understanding on 12 synthetic and realistic tasks averaging around 200K tokens, well beyond the roughly 10K tokens most earlier long-context benchmarks used."
  },
  {
   "aliases": [],
   "canonical_id": "instrumentaleval",
   "category": "safety",
   "disposition": "unassessed",
   "id": "instrumentaleval",
   "models_covered": 0,
   "name": "InstrumentalEval",
   "reasons": [],
   "summary": "InstrumentalEval presents agentic scenarios that create incentives for self-preservation, power-seeking or deception, then has a separate grader model judge whether the response pursued that instrumental goal."
  },
  {
   "aliases": [
    "BIG-bench intent_recognition",
    "SNIPS intent recognition (BIG-bench)"
   ],
   "canonical_id": "intent_recognition",
   "category": "domain",
   "disposition": "unassessed",
   "id": "intent_recognition",
   "models_covered": 0,
   "name": "Intent Recognition",
   "reasons": [],
   "summary": "A 693-item BIG-bench seven-way English intent task on SNIPS utterances, scored with multiple_choice_grade."
  },
  {
   "aliases": [
    "interactive_qa_mmlu",
    "HELM InteractiveQA MMLU"
   ],
   "canonical_id": "interactive_qa_mmlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "interactive_qa_mmlu",
   "models_covered": 0,
   "name": "InteractiveQA MMLU",
   "reasons": [],
   "summary": "HELM scenario that scores a small InteractiveQA subset of five MMLU subjects as four-choice exact match, not the full 57-subject test."
  },
  {
   "aliases": [
    "IPA NLI",
    "international phonetic alphabet nli"
   ],
   "canonical_id": "international_phonetic_alphabet_nli",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "international_phonetic_alphabet_nli",
   "models_covered": 0,
   "name": "IPA Natural Language Inference",
   "reasons": [],
   "summary": "A 126-item BIG-bench task that asks for entailment, contradiction or neutral on MultiNLI sentence pairs written in IPA."
  },
  {
   "aliases": [
    "IPA Transliterate",
    "international phonetic alphabet transliterate"
   ],
   "canonical_id": "international_phonetic_alphabet_transliterate",
   "category": "translation",
   "disposition": "unassessed",
   "id": "international_phonetic_alphabet_transliterate",
   "models_covered": 0,
   "name": "IPA Transliteration",
   "reasons": [],
   "summary": "A 1,003-example BIG-bench task that transliterates sentences between English and the International Phonetic Alphabet."
  },
  {
   "aliases": [
    "InternSandboxBenchmark",
    "internsandbox"
   ],
   "canonical_id": "internsandbox",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "internsandbox",
   "models_covered": 0,
   "name": "InternSandbox",
   "reasons": [],
   "summary": "OpenCompass generation eval over 80 named sandboxes, graded by intern_sandbox verifiers on local InternSandboxBenchmark_verified_V0.3.1 jsonl."
  },
  {
   "aliases": [
    "Intersection Points",
    "BIG-bench intersect_geometry"
   ],
   "canonical_id": "intersect_geometry",
   "category": "math",
   "disposition": "unassessed",
   "id": "intersect_geometry",
   "models_covered": 0,
   "name": "Intersect Geometry",
   "reasons": [],
   "summary": "A 250,000-item BIG-bench task that counts intersections among templated circles, polygons and segments, scored as 41-way multiple_choice_grade."
  },
  {
   "aliases": [
    "inverse_scaling_mc",
    "Inverse Scaling: When Bigger Isn't Better"
   ],
   "canonical_id": "inverse_scaling",
   "category": "composite",
   "disposition": "unassessed",
   "id": "inverse_scaling",
   "models_covered": 0,
   "name": "Inverse Scaling Prize",
   "reasons": [],
   "summary": "EleutherAI lm-eval packaging of Inverse Scaling Prize classification tasks, where larger models were observed to get worse, not better."
  },
  {
   "aliases": [
    "InverseIFEval"
   ],
   "canonical_id": "inverseifeval",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "inverseifeval",
   "models_covered": 0,
   "name": "Inverse IFEval",
   "reasons": [],
   "summary": "Inverse IFEval tests whether a model can override trained habits -- always answering, always correct, always-commented code -- to comply with instructions that deliberately conflict with them."
  },
  {
   "aliases": [
    "International Physics Olympiad 2025 Theoretical Examination",
    "IPhO 2025 Theoretical Exam"
   ],
   "canonical_id": "ipho_2025_theory",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "ipho_2025_theory",
   "models_covered": 1,
   "name": "IPhO 2025 Theory",
   "reasons": [],
   "summary": "The three-problem, 30-point theoretical examination of the 2025 International Physics Olympiad, used to test models against a fresh, human-graded physics exam."
  },
  {
   "aliases": [
    "CodeIPI",
    "ipi_coding_agent",
    "inspect_evals/ipi_coding_agent"
   ],
   "canonical_id": "ipi_coding_agent",
   "category": "safety",
   "disposition": "unassessed",
   "id": "ipi_coding_agent",
   "models_covered": 0,
   "name": "CodeIPI (Indirect Prompt Injection for Coding Agents)",
   "reasons": [],
   "summary": "Inspect Evals CodeIPI: 45 Docker coding-agent tasks that hide prompt injections in issues, comments, READMEs or configs, scored for resistance and bug-fix success."
  },
  {
   "aliases": [
    "Irony Identification"
   ],
   "canonical_id": "irony_identification",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "irony_identification",
   "models_covered": 0,
   "name": "Irony Identification (BIG-bench)",
   "reasons": [],
   "summary": "A 99-item BIG-bench binary task: label an English sentence as ironic or not ironic."
  },
  {
   "aliases": [
    "IWSLT 2017",
    "IWSLT2017",
    "iwslt2017-en-de"
   ],
   "canonical_id": "iwslt2017",
   "category": "translation",
   "disposition": "unassessed",
   "id": "iwslt2017",
   "models_covered": 0,
   "name": "IWSLT 2017 (OpenCompass English-German)",
   "reasons": [],
   "summary": "OpenCompass wrap of IWSLT 2017 TED English-to-German: generate German from English and score with sacreBLEU, using one BM25 in-context example."
  },
  {
   "aliases": [
    "ja_leaderboard"
   ],
   "canonical_id": "japanese_leaderboard",
   "category": "composite",
   "disposition": "unassessed",
   "id": "japanese_leaderboard",
   "models_covered": 0,
   "name": "Japanese Leaderboard (lm-evaluation-harness)",
   "reasons": [],
   "summary": "japanese_leaderboard is an lm-evaluation-harness group running eight independent Japanese NLP tasks -- JAQKET, four JGLUE tasks, MGSM, XL-Sum and XWinograd -- with no combined score computed."
  },
  {
   "aliases": [
    "JFinQA",
    "jfinqa: Japanese Financial Numerical Reasoning QA Benchmark"
   ],
   "canonical_id": "jfinqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "jfinqa",
   "models_covered": 0,
   "name": "jfinqa",
   "reasons": [],
   "summary": "jfinqa tests multi-step numerical reasoning over real Japanese corporate financial statements from EDINET filings, across three subtasks: calculation, internal-consistency checking and trend direction."
  },
  {
   "aliases": [
    "Jigsaw Multilingual Toxic Comment Classification",
    "jigsaw_multilingual"
   ],
   "canonical_id": "jigsawmultilingual",
   "category": "safety",
   "disposition": "unassessed",
   "id": "jigsawmultilingual",
   "models_covered": 0,
   "name": "Jigsaw Multilingual Toxic Comment Classification (OpenCompass)",
   "reasons": [],
   "summary": "OpenCompass wrap of Jigsaw's multilingual toxic-comment test files in six languages, scored with choice log-probs and AUC-ROC."
  },
  {
   "aliases": [
    "JSON Schema Bench"
   ],
   "canonical_id": "jsonschema_bench",
   "category": "generation",
   "disposition": "unassessed",
   "id": "jsonschema_bench",
   "models_covered": 0,
   "name": "JSONSchemaBench",
   "reasons": [],
   "summary": "JSONSchemaBench tests whether a model can generate JSON that both parses and validates against a supplied JSON Schema, drawn from real-world schemas across ten source domains."
  },
  {
   "aliases": [
    "K-Bench 01",
    "K-Bench01"
   ],
   "canonical_id": "k_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "k_bench",
   "models_covered": 0,
   "name": "K-Bench",
   "reasons": [],
   "summary": "K-Bench 01 scores nine frontier models on 178 private first-turn scientific requests from K-Dense Web using three LLM judges and an eight-dimension 0-10 rubric."
  },
  {
   "aliases": [],
   "canonical_id": "k_metbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "k_metbench",
   "models_covered": 0,
   "name": "K-MetBench",
   "reasons": [],
   "summary": "K-MetBench evaluates multimodal language models for Korean meteorological expertise across visual reasoning, logic, geo-cultural comprehension and domain analysis."
  },
  {
   "aliases": [
    "kanji_ascii_meaning",
    "kanji_ascii_pronunciation",
    "Kanji ASCII Art"
   ],
   "canonical_id": "kanji_ascii",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "kanji_ascii",
   "models_covered": 0,
   "name": "Kanji ASCII Art (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench pair of ASCII-art kanji tasks: guess on'yomi (904 items) or pick a KANJIDIC meaning (188 five-way items)."
  },
  {
   "aliases": [
    "Kannada Riddles",
    "kannada_json_task"
   ],
   "canonical_id": "kannada",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "kannada",
   "models_covered": 0,
   "name": "Kannada Riddles (BIG-bench)",
   "reasons": [],
   "summary": "A 316-item BIG-bench multiple-choice task that asks models to solve riddles written in Kannada."
  },
  {
   "aliases": [
    "KaoshiDataset"
   ],
   "canonical_id": "kaoshi",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "kaoshi",
   "models_covered": 0,
   "name": "Kaoshi (OpenCompass)",
   "reasons": [],
   "summary": "OpenCompass suite of Chinese professional-licence and kaoyan exam items in six formats, scored zero-shot with regex extraction; item count is not published."
  },
  {
   "aliases": [
    "Korean Benchmark for Legal Language Understanding"
   ],
   "canonical_id": "kbl",
   "category": "domain",
   "disposition": "unassessed",
   "id": "kbl",
   "models_covered": 0,
   "name": "KBL (Korean Benchmark for Legal Language Understanding)",
   "reasons": [],
   "summary": "Korean legal suite: 7 knowledge tasks, 4 reasoning tasks, and Korean bar-exam items, scored as letter exact match in lm-eval."
  },
  {
   "aliases": [
    "KLCE",
    "kcle_fix"
   ],
   "canonical_id": "kcle",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "kcle",
   "models_covered": 0,
   "name": "KCLE (OpenCompass)",
   "reasons": [],
   "summary": "OpenCompass multiple-choice set (abbrs kcle and kcle_fix) graded by an LLM judge; item count, licence and the acronym's expansion are not established."
  },
  {
   "aliases": [
    "KernelBench: Can LLMs Write Efficient GPU Kernels?"
   ],
   "canonical_id": "kernelbench",
   "category": "coding",
   "disposition": "unassessed",
   "id": "kernelbench",
   "models_covered": 0,
   "name": "KernelBench",
   "reasons": [],
   "summary": "A model rewrites PyTorch workloads as GPU kernels; fast_p scores the share that are both correct and at least p times faster than the PyTorch baseline."
  },
  {
   "aliases": [
    "key-value maps"
   ],
   "canonical_id": "key_value_maps",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "key_value_maps",
   "models_covered": 0,
   "name": "Key/Value Maps",
   "reasons": [],
   "summary": "101 Yes/No BIG-bench questions on whether formal statements about key/value maps hold."
  },
  {
   "aliases": [
    "Korean-MMLU",
    "k_mmlu"
   ],
   "canonical_id": "kmmlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "kmmlu",
   "models_covered": 0,
   "name": "KMMLU (Korean-MMLU)",
   "reasons": [],
   "summary": "35,030 Korean four-option exam questions across 45 subjects, collected from original Korean tests rather than translated from MMLU."
  },
  {
   "aliases": [
    "Known Unknowns (Hallucinations)"
   ],
   "canonical_id": "known_unknowns",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "known_unknowns",
   "models_covered": 0,
   "name": "Known Unknowns",
   "reasons": [],
   "summary": "46-item BIG-bench Lite set that pairs a known fact with a question whose honest answer is Unknown."
  },
  {
   "aliases": [
    "Koala Eval",
    "koala_test_set",
    "Koala test dataset"
   ],
   "canonical_id": "koala",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "koala",
   "models_covered": 0,
   "name": "Koala (HELM Instruct)",
   "reasons": [],
   "summary": "HELM Instruct scenario over 180 Koala user prompts, scored with a 1-5 Helpfulness critique rather than gold answers."
  },
  {
   "aliases": [
    "KOBEST",
    "KB-BoolQ",
    "KB-COPA",
    "KB-WiC",
    "KB-HellaSwag",
    "KB-SentiNeg"
   ],
   "canonical_id": "kobest",
   "category": "composite",
   "disposition": "unassessed",
   "id": "kobest",
   "models_covered": 0,
   "name": "KoBEST (Korean Balanced Evaluation of Significant Tasks)",
   "reasons": [],
   "summary": "Five human-written Korean NLU tasks (yes/no QA, causal alternatives, word sense, sentence completion, polarity under negation) scored as multiple-choice accuracy and macro F1."
  },
  {
   "aliases": [
    "KOR-Bench: Benchmarking Language Models on Knowledge-Orthogonal Reasoning Tasks"
   ],
   "canonical_id": "korbench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "korbench",
   "models_covered": 0,
   "name": "KOR-Bench",
   "reasons": [],
   "summary": "Tests applying an invented, never-seen rule across five categories; despite the id, this benchmark (KOR-Bench, \"Knowledge-Orthogonal Reasoning\") is English-language, not Korean."
  },
  {
   "aliases": [],
   "canonical_id": "kormedmcqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "kormedmcqa",
   "models_covered": 0,
   "name": "KorMedMCQA",
   "reasons": [],
   "summary": "7,469 five-option Korean multiple-choice questions from doctor, nurse, pharmacist and dentist licensing exams (2012-2024), with a recorded human-examinee average of 77.05%."
  },
  {
   "aliases": [
    "KPI EDGAR"
   ],
   "canonical_id": "kpi_edgar",
   "category": "domain",
   "disposition": "unassessed",
   "id": "kpi_edgar",
   "models_covered": 0,
   "name": "KPI-EDGAR",
   "reasons": [],
   "summary": "Sentence-level extraction of KPI names and values from SEC 10-K filings; an information-extraction task, not a reasoning benchmark, and HELM's version tests only a simplified 4-tag NER slice of the original 12-tag task."
  },
  {
   "aliases": [
    "Language Agent Biology Benchmark",
    "lab-bench"
   ],
   "canonical_id": "lab_bench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "lab_bench",
   "models_covered": 0,
   "name": "LAB-Bench",
   "reasons": [],
   "summary": "FutureHouse's eight-category, 2,457-question suite testing practical biology-research skills -- literature QA, figure/table reading, database and sequence work, and molecular cloning -- rather than textbook recall."
  },
  {
   "aliases": [
    "FigQA"
   ],
   "canonical_id": "lab_bench_figqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "lab_bench_figqa",
   "models_covered": 2,
   "name": "LAB-Bench: FigQA",
   "reasons": [],
   "summary": "226 multiple-choice questions that give a model only a biology-paper figure image, no caption or text, and ask it to reason about the figure's content."
  },
  {
   "aliases": [
    "FigQA with tools"
   ],
   "canonical_id": "lab_bench_figqa_tools",
   "category": "domain",
   "disposition": "unassessed",
   "id": "lab_bench_figqa_tools",
   "models_covered": 2,
   "name": "LAB-Bench: FigQA (with tools)",
   "reasons": [],
   "summary": "The same 226 LAB-Bench FigQA questions scored when a model has an image-cropping tool and extended reasoning available, instead of answering from a single look at the figure."
  },
  {
   "aliases": [
    "LAMBADA dataset",
    "Word prediction requiring a broad discourse context"
   ],
   "canonical_id": "lambada",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "lambada",
   "models_covered": 0,
   "name": "LAMBADA",
   "reasons": [],
   "summary": "A last-word-prediction test built from narrative passages that humans can only complete correctly after reading the whole passage, not just the final sentence."
  },
  {
   "aliases": [
    "lambada openai cloze",
    "lambada standard cloze"
   ],
   "canonical_id": "lambada_cloze",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "lambada_cloze",
   "models_covered": 0,
   "name": "LAMBADA Cloze",
   "reasons": [],
   "summary": "The LAMBADA last-word task with an explicit cloze blank in the prompt, scored as log-likelihood accuracy on the withheld word."
  },
  {
   "aliases": [
    "lambada_openai_mt",
    "lambada_mt"
   ],
   "canonical_id": "lambada_multilingual",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "lambada_multilingual",
   "models_covered": 0,
   "name": "LAMBADA multilingual (OpenAI MT)",
   "reasons": [],
   "summary": "lm-eval group of OpenAI-format LAMBADA last-word tests in English and machine-translated German, Spanish, French and Italian (5,153 passages each)."
  },
  {
   "aliases": [
    "lambada_openai_mt_stablelm",
    "lambada_mt_stablelm"
   ],
   "canonical_id": "lambada_multilingual_stablelm",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "lambada_multilingual_stablelm",
   "models_covered": 0,
   "name": "LAMBADA multilingual (Stable LM translations)",
   "reasons": [],
   "summary": "lm-eval group of OpenAI-format LAMBADA last-word tests using Stability AI's retranslations, not the older googletrans EleutherAI/lambada_openai files."
  },
  {
   "aliases": [
    "language games",
    "Pig Latin / Egg language (BIG-bench)"
   ],
   "canonical_id": "language_games",
   "category": "translation",
   "disposition": "unassessed",
   "id": "language_games",
   "models_covered": 0,
   "name": "Language Games (BIG-bench)",
   "reasons": [],
   "summary": "A 2,128-item BIG-bench suite that asks models to encode, decode, or answer in Pig Latin and Egg language."
  },
  {
   "aliases": [
    "BIG-bench language identification",
    "LID (BIG-bench)"
   ],
   "canonical_id": "language_identification",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "language_identification",
   "models_covered": 0,
   "name": "Language Identification (BIG-bench Lite)",
   "reasons": [],
   "summary": "A 10,000-item BIG-bench Lite task: name the language of a sentence, choosing among 11 labels drawn from 1,000 languages."
  },
  {
   "aliases": [],
   "canonical_id": "lawbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "lawbench",
   "models_covered": 0,
   "name": "LawBench",
   "reasons": [],
   "summary": "LawBench scores a model's Chinese legal knowledge across 20 tasks grouped into memorization, understanding and application, drawn from real legal databases, exams and court documents."
  },
  {
   "aliases": [
    "LCBench"
   ],
   "canonical_id": "lcbench",
   "category": "coding",
   "disposition": "unassessed",
   "id": "lcbench",
   "models_covered": 0,
   "name": "LCBench2023",
   "reasons": [],
   "summary": "OpenCompass's LCBench2023: 581 LeetCode weekly-contest problems (2022-2023) in English and Chinese, scored pass@1 by executing generated code against each problem's tests."
  },
  {
   "aliases": [
    "Large-scale Chinese Short Text Summarization"
   ],
   "canonical_id": "lcsts",
   "category": "generation",
   "disposition": "unassessed",
   "id": "lcsts",
   "models_covered": 0,
   "name": "LCSTS (Large-scale Chinese Short Text Summarization)",
   "reasons": [],
   "summary": "Chinese short-text summarization from Sina Weibo author summaries; OpenCompass scores the 725-pair test split with jieba-tokenized ROUGE."
  },
  {
   "aliases": [],
   "canonical_id": "leaderboard_dataset",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "leaderboard_dataset",
   "models_covered": 0,
   "name": "leaderboard-dataset",
   "reasons": [],
   "summary": "leaderboard-dataset is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "leaderboard_details",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "leaderboard_details",
   "models_covered": 0,
   "name": "leaderboard-details",
   "reasons": [],
   "summary": "leaderboard-details is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "leaderboard_requests",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "leaderboard_requests",
   "models_covered": 0,
   "name": "leaderboard-requests",
   "reasons": [],
   "summary": "leaderboard-requests is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "leaderboard_results",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "leaderboard_results",
   "models_covered": 0,
   "name": "leaderboard-results",
   "reasons": [],
   "summary": "leaderboard-results is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [
    "Plain English Summarization of Contracts"
   ],
   "canonical_id": "legal_contract_summarization",
   "category": "domain",
   "disposition": "unassessed",
   "id": "legal_contract_summarization",
   "models_covered": 0,
   "name": "Legal Contract Summarization (HELM)",
   "reasons": [],
   "summary": "HELM generation task: rewrite a short unilateral-contract snippet in plain English and score ROUGE-L against community summaries."
  },
  {
   "aliases": [
    "legal_opinion"
   ],
   "canonical_id": "legal_opinion_sentiment_classification",
   "category": "domain",
   "disposition": "unassessed",
   "id": "legal_opinion_sentiment_classification",
   "models_covered": 0,
   "name": "Legal Opinion Sentiment Classification (HELM)",
   "reasons": [],
   "summary": "HELM three-class task: label a legal-opinion phrase positive, negative, or neutral using Ratnayaka et al. OSF spreadsheets."
  },
  {
   "aliases": [
    "billsum_legal_summarization",
    "multilexsum_legal_summarization",
    "eurlexsum_legal_summarization"
   ],
   "canonical_id": "legal_summarization",
   "category": "domain",
   "disposition": "unassessed",
   "id": "legal_summarization",
   "models_covered": 0,
   "name": "Legal summarization (HELM)",
   "reasons": [],
   "summary": "HELM group that scores English summaries of US bills, US civil-rights case writeups, and EU acts with ROUGE-2."
  },
  {
   "aliases": [
    "legal support",
    "HELM LegalSupport"
   ],
   "canonical_id": "legal_support",
   "category": "domain",
   "disposition": "unassessed",
   "id": "legal_support",
   "models_covered": 0,
   "name": "LegalSupport",
   "reasons": [],
   "summary": "HELM binary legal task: choose which of two case parentheticals more strongly supports a passage mined from US opinions."
  },
  {
   "aliases": [
    "LEGALBENCH"
   ],
   "canonical_id": "legalbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "legalbench",
   "models_covered": 20,
   "name": "LegalBench",
   "reasons": [],
   "summary": "A collaboratively built suite of 162 tasks, contributed by lawyers and computer scientists, testing six categories of legal reasoning in language models."
  },
  {
   "aliases": [
    "L-Eval: Instituting Standardized Evaluation for Long Context Language Models"
   ],
   "canonical_id": "leval",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "leval",
   "models_covered": 0,
   "name": "L-Eval",
   "reasons": [],
   "summary": "20 closed- and open-ended subtasks over 508 long documents (3k-200k tokens); its length-instruction-enhanced protocol curbs n-gram metrics' bias toward longer outputs."
  },
  {
   "aliases": [
    "LexGLUE",
    "Legal GLUE"
   ],
   "canonical_id": "lex_glue",
   "category": "domain",
   "disposition": "unassessed",
   "id": "lex_glue",
   "models_covered": 0,
   "name": "LexGLUE (Legal General Language Understanding Evaluation)",
   "reasons": [],
   "summary": "English legal NLU suite of seven public datasets; HELM scores each subset as generation with classification_macro_f1 on test."
  },
  {
   "aliases": [
    "Lextreme",
    "Multilingual Legal Benchmark for Natural Language Understanding"
   ],
   "canonical_id": "lextreme",
   "category": "domain",
   "disposition": "unassessed",
   "id": "lextreme",
   "models_covered": 0,
   "name": "LEXTREME",
   "reasons": [],
   "summary": "Multilingual legal suite of 11 datasets / 18 HELM tasks; paper aggregate 61.3 for XLM-R large, later GitHub table higher for legal-adapted encoders."
  },
  {
   "aliases": [
    "Long Input Benchmark for Russian Analysis",
    "LIBRA Mini"
   ],
   "canonical_id": "libra",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "libra",
   "models_covered": 0,
   "name": "LIBRA (Long Input Benchmark for Russian Analysis)",
   "reasons": [],
   "summary": "Russian long-context suite of 18 tasks (21 in the 2024 paper), scored mainly with exact match from 4k up to 512k tokens."
  },
  {
   "aliases": [
    "LINGOLY",
    "Linguistic Olympiad Benchmark"
   ],
   "canonical_id": "lingoly",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "lingoly",
   "models_covered": 0,
   "name": "LingOly",
   "reasons": [],
   "summary": "1,133 UK-Linguistics-Olympiad-style puzzles across 90+ mostly low-resource languages, scored on direct accuracy and a no-context control that penalises memorisation."
  },
  {
   "aliases": [
    "BIG-bench linguistic_mappings"
   ],
   "canonical_id": "linguistic_mappings",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "linguistic_mappings",
   "models_covered": 0,
   "name": "Linguistic Mappings",
   "reasons": [],
   "summary": "A BIG-bench task that maps lemmas and sentences through tense, plural, questions, negation, and pronouns from a handful of shots."
  },
  {
   "aliases": [
    "BIG-bench linguistics_puzzles",
    "Linguistic Puzzles"
   ],
   "canonical_id": "linguistics_puzzles",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "linguistics_puzzles",
   "models_covered": 0,
   "name": "Linguistics Puzzles",
   "reasons": [],
   "summary": "A 2,000-item BIG-bench Lite task that induces a constructed language from paired sentences and translates one more sentence either way."
  },
  {
   "aliases": [
    "BIG-bench list_functions"
   ],
   "canonical_id": "list_functions",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "list_functions",
   "models_covered": 0,
   "name": "List Functions",
   "reasons": [],
   "summary": "A BIG-bench task that infers one of 250 functions on lists of natural numbers from a few input/output pairs and applies it to a new list."
  },
  {
   "aliases": [
    "LIT-RAGBench: Benchmarking Generator Capabilities of Large Language Models in Retrieval-Augmented Generation"
   ],
   "canonical_id": "lit_ragbench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "lit_ragbench",
   "models_covered": 0,
   "name": "LIT-RAGBench",
   "reasons": [],
   "summary": "LIT-RAGBench evaluates retrieval-augmented generation generators on 114 human-constructed Japanese questions and a curated English version across integration, reasoning, logic, table and abstention capabilities."
  },
  {
   "aliases": [
    "LCB",
    "livecodebench"
   ],
   "canonical_id": "live_code_bench",
   "category": "coding",
   "disposition": "unassessed",
   "id": "live_code_bench",
   "models_covered": 144,
   "name": "LiveCodeBench",
   "reasons": [],
   "summary": "LiveCodeBench scores code generation, self-repair, test-output prediction and code execution on dated competitive-programming problems, filterable by a model's training cutoff."
  },
  {
   "aliases": [
    "TREC-2017 LiveQA: Medical Question Answering Task",
    "LiveQA'17 Medical"
   ],
   "canonical_id": "live_qa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "live_qa",
   "models_covered": 0,
   "name": "LiveQA (TREC-2017 Medical Task)",
   "reasons": [],
   "summary": "104 real, historical (2017) consumer health questions from the TREC LiveQA medical track; despite the \"live\" name this is a static, fixed test set, not a continuously refreshed one."
  },
  {
   "aliases": [
    "Live Bench"
   ],
   "canonical_id": "livebench",
   "category": "composite",
   "disposition": "unverified",
   "id": "livebench",
   "models_covered": 0,
   "name": "LiveBench",
   "reasons": [
    "no qualifying current frontier/open coverage from different organizations",
    "review is not approved",
    "task, metric, and protocol are incomplete",
    "usefulness is not established",
    "usefulness is unknown"
   ],
   "summary": "A monthly-refreshed suite across math, coding, reasoning, data analysis, language and instruction following, graded by automatic ground truth rather than an LLM judge, to limit contamination."
  },
  {
   "aliases": [
    "LCB Pro"
   ],
   "canonical_id": "livecodebench_pro",
   "category": "coding",
   "disposition": "unassessed",
   "id": "livecodebench_pro",
   "models_covered": 0,
   "name": "LiveCodeBench Pro",
   "reasons": [],
   "summary": "LiveCodeBench Pro tests models on Olympiad-level Codeforces, ICPC and IOI problems, annotated by competitive-programming medalists, and still finds 0% pass@1 on hard problems for most models."
  },
  {
   "aliases": [],
   "canonical_id": "livemathbench",
   "category": "math",
   "disposition": "unassessed",
   "id": "livemathbench",
   "models_covered": 0,
   "name": "LiveMathBench",
   "reasons": [],
   "summary": "A bilingual, periodically re-released competition-math benchmark from recent AMC, CNMO, CCEE and Putnam problems, paired with G-Pass@k, a metric scoring correctness and stability across samples."
  },
  {
   "aliases": [],
   "canonical_id": "livereasonbench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "livereasonbench",
   "models_covered": 0,
   "name": "LiveReasonBench",
   "reasons": [],
   "summary": "An OpenCompass-maintained, free-response QA benchmark graded like OpenAI's SimpleQA and refreshed through periodic dated dataset versions; its underlying question set is not publicly downloadable."
  },
  {
   "aliases": [],
   "canonical_id": "livestembench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "livestembench",
   "models_covered": 0,
   "name": "LiveStemBench",
   "reasons": [],
   "summary": "An OpenCompass-maintained, Chinese-language STEM QA benchmark split into biology, chemistry and physics subsets, graded like OpenAI's SimpleQA; its question set is not publicly downloadable."
  },
  {
   "aliases": [
    "llm-compression",
    "Compression Represents Intelligence Linearly",
    "BPC compression corpora"
   ],
   "canonical_id": "llm_compression",
   "category": "generation",
   "disposition": "unassessed",
   "id": "llm_compression",
   "models_covered": 0,
   "name": "LLM Compression",
   "reasons": [],
   "summary": "Bits-per-character compression of three raw corpora, used as an unsupervised linear proxy for knowledge, coding, and math benchmarks."
  },
  {
   "aliases": [],
   "canonical_id": "llm_qbench",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "llm_qbench",
   "models_covered": 0,
   "name": "LLM-QBench",
   "reasons": [],
   "summary": "LLM-QBench is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "llm_soccerarena",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "llm_soccerarena",
   "models_covered": 0,
   "name": "LLM-SoccerArena",
   "reasons": [],
   "summary": "LLM-SoccerArena is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "llm_stats",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "llm_stats",
   "models_covered": 0,
   "name": "LLM Stats",
   "reasons": [],
   "summary": "LLM Stats is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [
    "ORT",
    "ORT Benchmark",
    "Out-of-Distribution Robustness Testing",
    "LLM Unlearning Should Be Form-Independent"
   ],
   "canonical_id": "llm_unlearning_should_be_form_independent",
   "category": "safety",
   "disposition": "unassessed",
   "id": "llm_unlearning_should_be_form_independent",
   "models_covered": 0,
   "name": "ORT (Out-of-Distribution Robustness Testing)",
   "reasons": [],
   "summary": "ORT checks whether an unlearning method erases knowledge of a real person across four query forms, not only the form used in the unlearning samples."
  },
  {
   "aliases": [
    "LMentry",
    "lm_entry",
    "HELM lm_entry"
   ],
   "canonical_id": "lm_entry",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "lm_entry",
   "models_covered": 0,
   "name": "LMentry",
   "reasons": [],
   "summary": "Twenty-five elementary English tasks that humans solve perfectly, scored as accuracy multiplied by robustness to trivial input changes."
  },
  {
   "aliases": [
    "Targeted Syntactic Evaluation of Language Models",
    "Marvin and Linzen 2018"
   ],
   "canonical_id": "lm_syneval",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "lm_syneval",
   "models_covered": 0,
   "name": "LM-SynEval (Targeted Syntactic Evaluation of Language Models)",
   "reasons": [],
   "summary": "72 auto-generated minimal-pair test sets probing whether a model's probabilities favour the grammatical sentence for subject-verb agreement, reflexive anaphora and negative polarity items."
  },
  {
   "aliases": [
    "Logic Grid Puzzles",
    "BIG-bench logic_grid_puzzle"
   ],
   "canonical_id": "logic_grid_puzzle",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "logic_grid_puzzle",
   "models_covered": 0,
   "name": "Logic Grid Puzzle",
   "reasons": [],
   "summary": "1,000 generated house-grid puzzles in BIG-bench; the model picks which house holds a queried trait from spatial clues."
  },
  {
   "aliases": [],
   "canonical_id": "logical_args",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "logical_args",
   "models_covered": 0,
   "name": "Logical Arguments",
   "reasons": [],
   "summary": "A 32-item BIG-bench task that asks which statement most strengthens or weakens a short argument, in five-option GRE style."
  },
  {
   "aliases": [],
   "canonical_id": "logical_deduction",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "logical_deduction",
   "models_covered": 0,
   "name": "Logical Deduction",
   "reasons": [],
   "summary": "1,500 BIG-bench questions that ask which object sits at a given position after a minimal set of ordering clues; also three 250-item BBH tasks."
  },
  {
   "aliases": [
    "Informal and Formal Fallacies",
    "BIG-bench logical_fallacy_detection"
   ],
   "canonical_id": "logical_fallacy_detection",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "logical_fallacy_detection",
   "models_covered": 0,
   "name": "Logical Fallacy Detection",
   "reasons": [],
   "summary": "2,800 BIG-bench items asking Valid vs Invalid on informal fallacies and categorical or linear syllogisms, under two prompts."
  },
  {
   "aliases": [
    "Sequential Order"
   ],
   "canonical_id": "logical_sequence",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "logical_sequence",
   "models_covered": 0,
   "name": "Logical Sequence",
   "reasons": [],
   "summary": "A 39-item BIG-bench task that asks which listed order of real-world events or entities is chronological; internally titled Sequential Order."
  },
  {
   "aliases": [],
   "canonical_id": "logiqa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "logiqa",
   "models_covered": 0,
   "name": "LogiQA",
   "reasons": [],
   "summary": "LogiQA scores multiple-choice logical reasoning questions taken from China's National Civil Servants Examination, translated into English by professional translators."
  },
  {
   "aliases": [
    "LogiQA2",
    "LogiQA2.0",
    "logiqa2_zh",
    "logiqa2_nli"
   ],
   "canonical_id": "logiqa2",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "logiqa2",
   "models_covered": 0,
   "name": "LogiQA 2.0",
   "reasons": [],
   "summary": "Expanded civil-service logical-reasoning suite: English and Chinese four-option MRC, plus a two-way NLI conversion of the same items."
  },
  {
   "aliases": [
    "Long Input Contexts",
    "BIG-bench long_context_integration"
   ],
   "canonical_id": "long_context_integration",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "long_context_integration",
   "models_covered": 0,
   "name": "Long Context Integration",
   "reasons": [],
   "summary": "A programmatic BIG-bench task that scores the log length of the longest social-graph context a model can search, count, or traverse."
  },
  {
   "aliases": [
    "LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding"
   ],
   "canonical_id": "longbench",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "longbench",
   "models_covered": 0,
   "name": "LongBench",
   "reasons": [],
   "summary": "A bilingual English/Chinese suite of 21 tasks across six categories, testing long-context understanding at moderate lengths of roughly 5k-15k words per document."
  },
  {
   "aliases": [
    "LongBench v2: Towards Deeper Understanding and Reasoning on Realistic Long-context Multitasks"
   ],
   "canonical_id": "longbenchv2",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "longbenchv2",
   "models_covered": 0,
   "name": "LongBench v2",
   "reasons": [],
   "summary": "503 hard multiple-choice questions with contexts from 8k to 2M words, built so a model must reason over long context rather than just retrieve, with human experts scoring only 53.7%."
  },
  {
   "aliases": [
    "Long Procedural Generation",
    "LongProc: Benchmarking Long-Context Language Models on Long Procedural Generation"
   ],
   "canonical_id": "longproc",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "longproc",
   "models_covered": 0,
   "name": "LongProc",
   "reasons": [],
   "summary": "Six procedural-generation tasks at 0.5K, 2K, and 8K output lengths that test whether long-context models can follow a procedure and emit a structured trace."
  },
  {
   "aliases": [
    "AR-LSAT",
    "LSAT Analytical Reasoning"
   ],
   "canonical_id": "lsat_qa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "lsat_qa",
   "models_covered": 0,
   "name": "LSAT (Analytical Reasoning)",
   "reasons": [],
   "summary": "A HELM scenario built from AR-LSAT: 2,091 five-option multiple-choice logic-puzzle questions from real 1991-2016 LSAT analytical-reasoning (\"logic games\") sections, testing constraint satisfaction."
  },
  {
   "aliases": [
    "LVEval",
    "LV-Eval: A Balanced Long-Context Benchmark with 5 Length Levels Up to 256K"
   ],
   "canonical_id": "lveval",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "lveval",
   "models_covered": 0,
   "name": "LV-Eval",
   "reasons": [],
   "summary": "A bilingual long-context QA benchmark with 11 datasets at five length levels from 16k to 256k words, built to fight knowledge leakage with confusing-fact insertion and keyword-recall metrics."
  },
  {
   "aliases": [
    "M3-Bench"
   ],
   "canonical_id": "m3_bench",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "m3_bench",
   "models_covered": 0,
   "name": "M3-BENCH",
   "reasons": [],
   "summary": "M3-BENCH evaluates LLM agent social behavior in 24 mixed-motive games using behavioral, reasoning-process, and communication views."
  },
  {
   "aliases": [],
   "canonical_id": "m3_duplexbench",
   "category": "generation",
   "disposition": "unassessed",
   "id": "m3_duplexbench",
   "models_covered": 0,
   "name": "M3-DuplexBench",
   "reasons": [],
   "summary": "M3-DuplexBench evaluates multilingual full-duplex spoken dialogue with turn-taking, backchannels, and user interruptions."
  },
  {
   "aliases": [
    "MaCBench: Probing the limitations of multimodal language models for chemistry and materials research"
   ],
   "canonical_id": "macbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "macbench",
   "models_covered": 0,
   "name": "MaCBench",
   "reasons": [],
   "summary": "A 1,153-question, 34-subset benchmark testing whether vision-language models can extract, reason about and interpret chemistry and materials-science data from images paired with text."
  },
  {
   "aliases": [],
   "canonical_id": "madinah_qa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "madinah_qa",
   "models_covered": 0,
   "name": "MadinahQA",
   "reasons": [],
   "summary": "983 Arabic multiple-choice questions on Arabic language and grammar, an exact two-subject slice of ArabicMMLU repackaged standalone; now tracked in the OALL v2 Arabic leaderboard."
  },
  {
   "aliases": [
    "MakeMePay",
    "make-me-pay"
   ],
   "canonical_id": "make_me_pay",
   "category": "safety",
   "disposition": "unassessed",
   "id": "make_me_pay",
   "models_covered": 0,
   "name": "Make Me Pay",
   "reasons": [],
   "summary": "A two-model chat where a con-artist tries to make a mark holding $100 type a donation tag; used as a persuasion and manipulation eval."
  },
  {
   "aliases": [
    "Make Me Say",
    "make-me-say",
    "make_me_say"
   ],
   "canonical_id": "makemesay",
   "category": "safety",
   "disposition": "unassessed",
   "id": "makemesay",
   "models_covered": 0,
   "name": "MakeMeSay",
   "reasons": [],
   "summary": "A 30-turn two-model game where a manipulator tries to make a naive partner say a secret codeword without saying it or being guessed."
  },
  {
   "aliases": [
    "MAS-Bench",
    "MAS-Bench-Eval"
   ],
   "canonical_id": "mas_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "mas_bench",
   "models_covered": 0,
   "name": "MAS-Bench",
   "reasons": [],
   "summary": "MAS-Bench scores Android GUI agents on 139 live-app tasks when they may call APIs, deep links, and RPA scripts instead of tapping through every screen."
  },
  {
   "aliases": [
    "MASBench",
    "MasBench",
    "MASBENCH",
    "MAS-Orchestra"
   ],
   "canonical_id": "mas_orchestra",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "mas_orchestra",
   "models_covered": 0,
   "name": "MASBench",
   "reasons": [],
   "summary": "MASBench varies synthetic task graphs on five axes so a run can show when a multi-agent system beats a single agent, rather than quoting one headline score."
  },
  {
   "aliases": [
    "MASK",
    "MASK Benchmark",
    "cais/MASK",
    "inspect_evals/mask"
   ],
   "canonical_id": "mask",
   "category": "safety",
   "disposition": "unassessed",
   "id": "mask",
   "models_covered": 0,
   "name": "MASK (Model Alignment between Statements and Knowledge)",
   "reasons": [],
   "summary": "MASK tests whether a model contradicts its own beliefs when pressured to lie, scoring honesty separately from factual accuracy."
  },
  {
   "aliases": [
    "Mastermath2024v1",
    "mastermath2024v1_gen"
   ],
   "canonical_id": "mastermath2024v1",
   "category": "math",
   "disposition": "unassessed",
   "id": "mastermath2024v1",
   "models_covered": 0,
   "name": "Mastermath2024v1 (OpenCompass)",
   "reasons": [],
   "summary": "OpenCompass four-option Chinese kaoyan math MCQ loader; item count is not published in the configs."
  },
  {
   "aliases": [
    "MastermindEval",
    "mastermind_easy",
    "mastermind_hard",
    "mastermind_24_easy",
    "mastermind_24_hard",
    "mastermind_35_easy",
    "mastermind_35_hard",
    "mastermind_46_easy",
    "mastermind_46_hard"
   ],
   "canonical_id": "mastermind",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "mastermind",
   "models_covered": 0,
   "name": "MastermindEval",
   "reasons": [],
   "summary": "lm-eval's MastermindEval tag: six four-way MC tasks that ask for the last remaining Mastermind code after Knuth-style hints."
  },
  {
   "aliases": [],
   "canonical_id": "matbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "matbench",
   "models_covered": 0,
   "name": "Matbench",
   "reasons": [],
   "summary": "A 13-task materials-property-prediction suite built for classical ML; OpenCompass's LLM harness adapts its 4 composition-only tasks into text prompts graded by regex or an LLM judge."
  },
  {
   "aliases": [
    "Hendrycks MATH",
    "MATH dataset",
    "hendrycks_math",
    "minerva_math"
   ],
   "canonical_id": "math",
   "category": "math",
   "disposition": "unassessed",
   "id": "math",
   "models_covered": 0,
   "name": "MATH (Mathematics Aptitude Test of Heuristics)",
   "reasons": [],
   "summary": "The original 12,500-problem competition-mathematics benchmark from Hendrycks et al. 2021; its 5,000-problem test split is the parent of the smaller MATH-500 subset most current model cards actually report."
  },
  {
   "aliases": [
    "MATH 401",
    "MATH401",
    "math401-llm"
   ],
   "canonical_id": "math401",
   "category": "math",
   "disposition": "unassessed",
   "id": "math401",
   "models_covered": 0,
   "name": "MATH 401",
   "reasons": [],
   "summary": "401 constructed arithmetic expressions used to test LLM calculation; OpenCompass runs a four-shot English cloze with 1e-3 tolerance."
  },
  {
   "aliases": [
    "MATH500",
    "hendrycks_math500",
    "minerva_math500"
   ],
   "canonical_id": "math_500",
   "category": "math",
   "disposition": "unassessed",
   "id": "math_500",
   "models_covered": 354,
   "name": "MATH-500",
   "reasons": [],
   "summary": "A fixed 500-problem subset of the MATH test set used to grade free-response competition mathematics after the rest of the split was moved into training data."
  },
  {
   "aliases": [],
   "canonical_id": "mathbench",
   "category": "math",
   "disposition": "unassessed",
   "id": "mathbench",
   "models_covered": 0,
   "name": "MathBench",
   "reasons": [],
   "summary": "A bilingual, 3,709-problem suite spanning five education stages from arithmetic to college, each scored separately on theory recall and applied problem-solving using circular multiple-choice evaluation."
  },
  {
   "aliases": [
    "Mathematical Induction",
    "BIG-bench mathematical_induction"
   ],
   "canonical_id": "mathematical_induction",
   "category": "math",
   "disposition": "unassessed",
   "id": "mathematical_induction",
   "models_covered": 0,
   "name": "Mathematical Induction (BIG-bench)",
   "reasons": [],
   "summary": "A 70-item BIG-bench yes/no task: is this short induction argument structurally valid, even if a premise is false?"
  },
  {
   "aliases": [],
   "canonical_id": "mathqa",
   "category": "math",
   "disposition": "unassessed",
   "id": "mathqa",
   "models_covered": 0,
   "name": "MathQA",
   "reasons": [],
   "summary": "A 37k-problem multiple-choice math word problem dataset built by re-annotating AQuA-RAT with interpretable, formal operation programs rather than free-text rationales."
  },
  {
   "aliases": [],
   "canonical_id": "mathvista",
   "category": "math",
   "disposition": "unassessed",
   "id": "mathvista",
   "models_covered": 66,
   "name": "MathVista",
   "reasons": [],
   "summary": "6,141 examples testing mathematical reasoning across charts, diagrams, word problems and textbook figures, pooled from 28 existing datasets plus three new ones."
  },
  {
   "aliases": [
    "Matrix Shapes",
    "BIG-bench matrixshapes"
   ],
   "canonical_id": "matrixshapes",
   "category": "math",
   "disposition": "unassessed",
   "id": "matrixshapes",
   "models_covered": 0,
   "name": "Matrix Shapes (BIG-bench)",
   "reasons": [],
   "summary": "A 5,000-item BIG-bench task: emit the result shape of a 2\u20135 step chain of matrix operations."
  },
  {
   "aliases": [
    "Mostly Basic Python Problems"
   ],
   "canonical_id": "mbpp",
   "category": "coding",
   "disposition": "unassessed",
   "id": "mbpp",
   "models_covered": 0,
   "name": "MBPP (Mostly Basic Python Problems)",
   "reasons": [],
   "summary": "Crowd-sourced, entry-level Python programming problems checked by unit tests; reported numbers vary widely because at least three differently-sized versions of the dataset are in circulation."
  },
  {
   "aliases": [
    "MBPP CN",
    "MBPP Chinese"
   ],
   "canonical_id": "mbpp_cn",
   "category": "coding",
   "disposition": "unassessed",
   "id": "mbpp_cn",
   "models_covered": 0,
   "name": "MBPP-CN",
   "reasons": [],
   "summary": "OpenCompass's Chinese-instruction variant of MBPP: the same unit-tested Python tasks, prompted in Chinese rather than English."
  },
  {
   "aliases": [
    "MBPP Plus",
    "mbppplus",
    "EvalPlus MBPP+"
   ],
   "canonical_id": "mbpp_plus",
   "category": "coding",
   "disposition": "unassessed",
   "id": "mbpp_plus",
   "models_covered": 0,
   "name": "MBPP+",
   "reasons": [],
   "summary": "EvalPlus's stricter MBPP: the same crowd-sourced Python tasks, filtered and graded against about 35 times more tests so fragile completions fail."
  },
  {
   "aliases": [
    "MBPP-Pro",
    "CodeEval-Pro MBPP"
   ],
   "canonical_id": "mbpp_pro",
   "category": "coding",
   "disposition": "unassessed",
   "id": "mbpp_pro",
   "models_covered": 0,
   "name": "MBPP Pro",
   "reasons": [],
   "summary": "CodeEval-Pro's harder MBPP: each base problem is paired with a second task the model must solve by calling its own solution to the first."
  },
  {
   "aliases": [
    "MC-TACO",
    "MCTACO",
    "mc-taco",
    "Multiple Choice TemporAl COmmonsense"
   ],
   "canonical_id": "mc_taco",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "mc_taco",
   "models_covered": 0,
   "name": "MC-TACO",
   "reasons": [],
   "summary": "13k English temporal-commonsense candidate answers; the paper scores question-level EM/F1, while lm-eval reports pair-level acc/F1."
  },
  {
   "aliases": [
    "MCP-Bench",
    "MCPBench",
    "mcp-bench"
   ],
   "canonical_id": "mcp_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "mcp_bench",
   "models_covered": 0,
   "name": "MCP-Bench",
   "reasons": [],
   "summary": "MCP-Bench scores agents on 104 fuzzy multi-step tasks that must call tools across 28 live MCP servers (250 tools) without being told the tool names."
  },
  {
   "aliases": [
    "MedConceptsQA",
    "medconceptsqa"
   ],
   "canonical_id": "med_concepts_qa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "med_concepts_qa",
   "models_covered": 0,
   "name": "MedConceptsQA",
   "reasons": [],
   "summary": "Four-option questions that ask a model to pick the correct description of an ICD or ATC medical code among related distractors, across three difficulty levels."
  },
  {
   "aliases": [],
   "canonical_id": "med_dialog",
   "category": "domain",
   "disposition": "unassessed",
   "id": "med_dialog",
   "models_covered": 0,
   "name": "MedDialog",
   "reasons": [],
   "summary": "MedDialog evaluates concise summaries of English doctor-patient conversations from HealthCareMagic and iCliniq."
  },
  {
   "aliases": [
    "MedParaSimplification"
   ],
   "canonical_id": "med_paragraph_simplification",
   "category": "generation",
   "disposition": "unassessed",
   "id": "med_paragraph_simplification",
   "models_covered": 0,
   "name": "Paragraph-level Simplification of Medical Texts",
   "reasons": [],
   "summary": "This benchmark maps technical medical abstracts to lay-language summaries across clinical topics."
  },
  {
   "aliases": [],
   "canonical_id": "medalign",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "medalign",
   "models_covered": 0,
   "name": "MedAlign",
   "reasons": [],
   "summary": "MedAlign tests instruction following grounded in longitudinal electronic health records and clinician responses."
  },
  {
   "aliases": [
    "MedBench: A Comprehensive, Standardized, and Reliable Benchmarking System for Evaluating Chinese Medical Large Language Models"
   ],
   "canonical_id": "medbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medbench",
   "models_covered": 0,
   "name": "MedBench",
   "reasons": [],
   "summary": "OpenCompass's own cloud-hosted, actively versioned Chinese medical benchmark (v1-v5), covering LLM, multimodal and clinical-agent tracks; documented here as distinct from an unrelated, same-named 2023 benchmark."
  },
  {
   "aliases": [
    "MedBullets",
    "MedBullet"
   ],
   "canonical_id": "medbullets",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medbullets",
   "models_covered": 0,
   "name": "Medbullets",
   "reasons": [],
   "summary": "308 USMLE Step 2/3-style clinical vignette questions with expert explanations, built specifically to be harder than MedQA and to test explanation quality, not just the answer letter."
  },
  {
   "aliases": [
    "MedCalc_Bench"
   ],
   "canonical_id": "medcalc_bench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medcalc_bench",
   "models_covered": 0,
   "name": "MedCalc-Bench",
   "reasons": [],
   "summary": "Tests whether a model can compute clinical values (dosages, risk scores, dates) from a patient note the way a bedside medical calculator would, graded by exact match or numeric tolerance."
  },
  {
   "aliases": [],
   "canonical_id": "medec",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medec",
   "models_covered": 0,
   "name": "MEDEC",
   "reasons": [],
   "summary": "MEDEC evaluates detection and correction of medical errors in clinical narratives."
  },
  {
   "aliases": [],
   "canonical_id": "medhallu",
   "category": "safety",
   "disposition": "unassessed",
   "id": "medhallu",
   "models_covered": 0,
   "name": "MedHallu",
   "reasons": [],
   "summary": "MedHallu classifies whether biomedical answers grounded in PubMed knowledge are factual or hallucinated."
  },
  {
   "aliases": [
    "MedHELM"
   ],
   "canonical_id": "medhelm_configurable",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medhelm_configurable",
   "models_covered": 0,
   "name": "MedHELM Configurable",
   "reasons": [],
   "summary": "MedHELM Configurable is a HELM scenario wrapper for configurable biomedical datasets, prompts, references, and metrics."
  },
  {
   "aliases": [
    "MEDIQA",
    "MEDIQA-QA",
    "MEDIQA 2019 QA",
    "mediqa_qa"
   ],
   "canonical_id": "medi_qa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medi_qa",
   "models_covered": 0,
   "name": "MEDIQA (HELM)",
   "reasons": [],
   "summary": "HELM's generation wrap of MEDIQA 2019 Task 3, scoring a free-form answer to a consumer health question with an LLM jury against the expert-ranked gold answer."
  },
  {
   "aliases": [],
   "canonical_id": "medical_questions_russian",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medical_questions_russian",
   "models_covered": 0,
   "name": "Medical Questions Russian",
   "reasons": [],
   "summary": "BIG-bench Medical Questions Russian tests yes/no comprehension of Russian medical text."
  },
  {
   "aliases": [
    "Medication_QA",
    "Medication QA MedInfo 2019",
    "MedInfo2019-QA-Medications"
   ],
   "canonical_id": "medication_qa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medication_qa",
   "models_covered": 0,
   "name": "MedicationQA",
   "reasons": [],
   "summary": "Open-ended consumer questions about medications paired with trusted reference answers drawn from DailyMed, MedlinePlus and similar sources."
  },
  {
   "aliases": [
    "mediqa_qa2019_perplexity",
    "MEDIQA-QA 2019 (lm-eval)"
   ],
   "canonical_id": "mediqa_qa2019",
   "category": "domain",
   "disposition": "unassessed",
   "id": "mediqa_qa2019",
   "models_covered": 0,
   "name": "MEDIQA 2019 QA (lm-eval)",
   "reasons": [],
   "summary": "lm-eval's generation wrap of MEDIQA 2019 Task 3: write an English answer to a consumer health question and score overlap against the first listed gold answer."
  },
  {
   "aliases": [
    "Med-MCQA"
   ],
   "canonical_id": "medmcqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medmcqa",
   "models_covered": 4,
   "name": "MedMCQA",
   "reasons": [],
   "summary": "Over 194,000 multiple-choice questions from India's AIIMS and NEET PG medical entrance exams, across 21 subjects."
  },
  {
   "aliases": [
    "MedQA-USMLE",
    "MedQA-USMLE-4-options"
   ],
   "canonical_id": "medqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medqa",
   "models_covered": 51,
   "name": "MedQA",
   "reasons": [],
   "summary": "Four-option USMLE-style clinical multiple-choice questions, the most widely reported medical exam benchmark for LLMs."
  },
  {
   "aliases": [
    "medtext_perplexity",
    "BI55/MedText"
   ],
   "canonical_id": "medtext",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medtext",
   "models_covered": 0,
   "name": "MedText (lm-eval)",
   "reasons": [],
   "summary": "lm-eval's generation wrap of BI55/MedText: write a diagnosis and treatment plan from an English patient presentation and score overlap metrics."
  },
  {
   "aliases": [
    "MedXpertQA",
    "MedXpertQA-Text",
    "MedXpertQA Text"
   ],
   "canonical_id": "medxpertqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medxpertqa",
   "models_covered": 0,
   "name": "MedXpertQA Text",
   "reasons": [],
   "summary": "The text-only split of MedXpertQA, ten-option expert-level clinical questions across 17 specialties, which is what OpenCompass registers as MedXpertQA."
  },
  {
   "aliases": [
    "MedXpertQA Multimodal",
    "MedXpertQA-MM"
   ],
   "canonical_id": "medxpertqa_multimodal",
   "category": "domain",
   "disposition": "unassessed",
   "id": "medxpertqa_multimodal",
   "models_covered": 1,
   "name": "MedXpertQA MM",
   "reasons": [],
   "summary": "The multimodal split of MedXpertQA, pairing expert-level clinical questions with medical images and up to ten answer options."
  },
  {
   "aliases": [
    "melt_information_retrieval",
    "melt_information_retrieval_mmarco",
    "melt_information_retrieval_mrobust"
   ],
   "canonical_id": "melt_ir",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "melt_ir",
   "models_covered": 0,
   "name": "HELM MELT information retrieval (Vietnamese mMARCO and mRobust)",
   "reasons": [],
   "summary": "HELM's Vietnamese information-retrieval track: rank passages for a query on translated mMARCO (RR@10) and mRobust (NDCG@10)."
  },
  {
   "aliases": [
    "melt_knowledge_zalo",
    "melt_knowledge_vimmrc",
    "ZaloE2E",
    "ViMMRC"
   ],
   "canonical_id": "melt_knowledge",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "melt_knowledge",
   "models_covered": 0,
   "name": "HELM MELT knowledge (ZaloE2E and ViMMRC)",
   "reasons": [],
   "summary": "HELM's Vietnamese knowledge track: closed-book answers on ZaloE2E and multiple-choice reading on ViMMRC, both scored by quasi-exact match."
  },
  {
   "aliases": [
    "melt_synthetic_reasoning_natural",
    "MELT SRN"
   ],
   "canonical_id": "melt_srn",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "melt_srn",
   "models_covered": 0,
   "name": "HELM MELT synthetic reasoning (natural language)",
   "reasons": [],
   "summary": "HELM's Vietnamese synthetic-reasoning (natural) task: deduce attributes from generated Vietnamese rules and facts, scored by set-overlap F1."
  },
  {
   "aliases": [
    "melt_synthetic_reasoning_pattern_match",
    "melt_synthetic_reasoning_variable_substitution",
    "melt_synthetic_reasoning_induction"
   ],
   "canonical_id": "melt_synthetic_reasoning",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "melt_synthetic_reasoning",
   "models_covered": 0,
   "name": "HELM MELT synthetic reasoning (abstract symbols)",
   "reasons": [],
   "summary": "HELM MELT's Vietnamese LIME-style tasks: match a pattern, substitute variables, or induce a rule over abstract symbols filled with Vietnamese words."
  },
  {
   "aliases": [
    "melt_translation_opus100",
    "melt_translation_phomt",
    "MELT OPUS100",
    "MELT PhoMT"
   ],
   "canonical_id": "melt_translation",
   "category": "translation",
   "disposition": "unassessed",
   "id": "melt_translation",
   "models_covered": 0,
   "name": "MELT translation (HELM Vietnamese OPUS-100 and PhoMT)",
   "reasons": [],
   "summary": "HELM's Vietnamese\u2013English translation pair of OPUS-100 and PhoMT, scored mainly by quasi-exact match rather than BLEU."
  },
  {
   "aliases": [
    "MedHELM MentalHealth",
    "mental_health_accuracy"
   ],
   "canonical_id": "mental_health",
   "category": "domain",
   "disposition": "unassessed",
   "id": "mental_health",
   "models_covered": 0,
   "name": "MentalHealth (MedHELM)",
   "reasons": [],
   "summary": "MedHELM's private counseling task: generate the next English counselor turn from dialogue history and score it with an LLM jury."
  },
  {
   "aliases": [
    "Medical Question Summarization",
    "Consumer Health Question Summarization"
   ],
   "canonical_id": "meqsum",
   "category": "generation",
   "disposition": "unassessed",
   "id": "meqsum",
   "models_covered": 0,
   "name": "MeQSum",
   "reasons": [],
   "summary": "1,000 real consumer health questions paired with an expert-condensed one-sentence summary, from the ACL 2019 paper that introduced medical question summarisation as a task."
  },
  {
   "aliases": [
    "MetaBench",
    "metabench-A"
   ],
   "canonical_id": "metabench",
   "category": "composite",
   "disposition": "unassessed",
   "id": "metabench",
   "models_covered": 0,
   "name": "metabench",
   "reasons": [],
   "summary": "An 858-item sparse subsample of ARC, GSM8K, HellaSwag, MMLU, TruthfulQA and WinoGrande that reconstructs Open LLM Leaderboard scores from a few percent of the items."
  },
  {
   "aliases": [
    "metaphor boolean"
   ],
   "canonical_id": "metaphor_boolean",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "metaphor_boolean",
   "models_covered": 0,
   "name": "Metaphor Boolean (BIG-bench)",
   "reasons": [],
   "summary": "A 680-item BIG-bench True/False task: is the second sentence a correct reading of a metaphor."
  },
  {
   "aliases": [
    "met2lit / lit2met"
   ],
   "canonical_id": "metaphor_understanding",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "metaphor_understanding",
   "models_covered": 0,
   "name": "Metaphor Understanding (BIG-bench)",
   "reasons": [],
   "summary": "A 234-item BIG-bench four-way task mapping metaphors to literal paraphrases and the reverse."
  },
  {
   "aliases": [
    "Multilingual Grade School Math Benchmark",
    "Multilingual GSM8K"
   ],
   "canonical_id": "mgsm",
   "category": "math",
   "disposition": "unassessed",
   "id": "mgsm",
   "models_covered": 45,
   "name": "MGSM (Multilingual Grade School Math)",
   "reasons": [],
   "summary": "The same 250 GSM8K grade-school math problems, human-translated into ten languages, to test whether chain-of-thought reasoning holds up outside English."
  },
  {
   "aliases": [
    "MIMIC-IV-BHC",
    "MIMIC-IV-Ext-BHC",
    "Brief Hospital Course"
   ],
   "canonical_id": "mimic_bhc",
   "category": "domain",
   "disposition": "unassessed",
   "id": "mimic_bhc",
   "models_covered": 0,
   "name": "MIMIC-BHC (MedHELM)",
   "reasons": [],
   "summary": "MedHELM's gated wrap of MIMIC-IV-BHC: write a Brief Hospital Course from a discharge note, scored by an LLM jury plus overlap metrics."
  },
  {
   "aliases": [],
   "canonical_id": "mimic_repsum",
   "category": "domain",
   "disposition": "unassessed",
   "id": "mimic_repsum",
   "models_covered": 0,
   "name": "MIMIC-III Report Summarization (lm-eval)",
   "reasons": [],
   "summary": "lm-eval task mimic_repsum: write an Impression from Findings parsed out of a MIMIC-III hospital-course dump, scored with ROUGE, BLEU, BERTScore, BLEURT and RadGraph-F1."
  },
  {
   "aliases": [
    "MIMIC RRS",
    "Radiology Report Summarization"
   ],
   "canonical_id": "mimic_rrs",
   "category": "domain",
   "disposition": "unassessed",
   "id": "mimic_rrs",
   "models_covered": 0,
   "name": "MIMIC-RRS (MedHELM)",
   "reasons": [],
   "summary": "MedHELM's gated wrap of MIMIC-RRS on MIMIC-III: generate an Impression from Findings, scored by an LLM jury plus overlap metrics."
  },
  {
   "aliases": [
    "MIMIC-IV Billing Code",
    "mimiciv_icd10"
   ],
   "canonical_id": "mimiciv_billing_code",
   "category": "domain",
   "disposition": "unassessed",
   "id": "mimiciv_billing_code",
   "models_covered": 0,
   "name": "MIMIC-IV Billing Code (MedHELM)",
   "reasons": [],
   "summary": "MedHELM's gated MIMIC-IV task: extract ICD-10 codes from an English discharge note and score micro-F1 against gold codes."
  },
  {
   "aliases": [
    "Mind 2 Web",
    "MindAct"
   ],
   "canonical_id": "mind2web",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "mind2web",
   "models_covered": 0,
   "name": "Mind2Web",
   "reasons": [],
   "summary": "A web-agent benchmark of 2,000-plus real-website tasks; the model picks the next HTML element and action (click, type, or select) on three held-out splits."
  },
  {
   "aliases": [
    "Mind2Web SC",
    "Mind2Web Safety Control"
   ],
   "canonical_id": "mind2web_sc",
   "category": "safety",
   "disposition": "unassessed",
   "id": "mind2web_sc",
   "models_covered": 0,
   "name": "Mind2Web-SC",
   "reasons": [],
   "summary": "A GuardAgent safety eval: generate and execute code that grants or denies a SeeAct web action under six user-constraint rules."
  },
  {
   "aliases": [
    "MinuteMysteriesQA",
    "minute mysteries"
   ],
   "canonical_id": "minute_mysteries_qa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "minute_mysteries_qa",
   "models_covered": 0,
   "name": "Minute Mysteries QA",
   "reasons": [],
   "summary": "A BIG-bench reading-comprehension task that asks a model, given a short crime story, to identify the perpetrator and explain the clues that support that deduction."
  },
  {
   "aliases": [
    "Making a MIRACL",
    "Multilingual Information Retrieval Across a Continuum of Languages",
    "MIRACL: A Multilingual Retrieval Dataset Covering 18 Diverse Languages"
   ],
   "canonical_id": "miracl",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "miracl",
   "models_covered": 90,
   "name": "MIRACL",
   "reasons": [],
   "summary": "Monolingual ad hoc retrieval over Wikipedia passages in 18 languages, built from native-speaker queries and relevance judgments."
  },
  {
   "aliases": [],
   "canonical_id": "misconceptions",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "misconceptions",
   "models_covered": 0,
   "name": "Misconceptions",
   "reasons": [],
   "summary": "A 219-item BIG-bench true/false task that asks whether a short English statement is a popular misconception or a fact."
  },
  {
   "aliases": [],
   "canonical_id": "misconceptions_russian",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "misconceptions_russian",
   "models_covered": 0,
   "name": "Misconceptions (Russian)",
   "reasons": [],
   "summary": "A 49-item BIG-bench Lite task that scores whether a model assigns higher probability to a true Russian statement than to a matched misconception."
  },
  {
   "aliases": [
    "mle-bench"
   ],
   "canonical_id": "mle_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "mle_bench",
   "models_covered": 0,
   "name": "MLE-bench",
   "reasons": [],
   "summary": "Tests whether an AI agent can act as a machine learning engineer on 75 real Kaggle competitions, graded against the competitions' own medal thresholds."
  },
  {
   "aliases": [
    "MultiLingual Question Answering",
    "MLQA (MultiLingual Question Answering)"
   ],
   "canonical_id": "mlqa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mlqa",
   "models_covered": 0,
   "name": "MLQA",
   "reasons": [],
   "summary": "Parallel extractive question answering in seven languages, testing whether a model can recover an answer span from Wikipedia even when the question is in a different language."
  },
  {
   "aliases": [
    "MLRC-Bench: Can Language Agents Solve Machine Learning Research Challenges?"
   ],
   "canonical_id": "mlrc_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "mlrc_bench",
   "models_covered": 0,
   "name": "MLRC-Bench",
   "reasons": [],
   "summary": "Tests whether a language agent can propose and implement a genuinely novel ML method across 7 real research-competition tasks, scored against each competition's own baseline and top human result."
  },
  {
   "aliases": [
    "MMBench_DEV_EN",
    "MMBench-EN"
   ],
   "canonical_id": "mmbench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "mmbench",
   "models_covered": 0,
   "name": "MMBench",
   "reasons": [],
   "summary": "A bilingual vision-language multiple-choice benchmark with CircularEval over 20 ability dimensions; English and Chinese Dev and Test splits."
  },
  {
   "aliases": [
    "Multimodal Multi-image Understanding",
    "MMIU-Benchmark"
   ],
   "canonical_id": "mmiu",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "mmiu",
   "models_covered": 0,
   "name": "MMIU",
   "reasons": [],
   "summary": "A 11,698-question multi-image multiple-choice benchmark spanning 52 tasks and 7 image-relationship types, scored by accuracy."
  },
  {
   "aliases": [
    "Massive Multitask Language Understanding",
    "Hendrycks Test"
   ],
   "canonical_id": "mmlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu",
   "models_covered": 0,
   "name": "MMLU (Massive Multitask Language Understanding)",
   "reasons": [],
   "summary": "A 57-subject, four-choice knowledge test from elementary to professional difficulty; the standard reference for broad model knowledge since 2020, now saturated at the frontier."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_abstract_algebra",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_abstract_algebra",
   "models_covered": 50,
   "name": "MMLU: Abstract Algebra",
   "reasons": [],
   "summary": "MMLU subject subset: Undergraduate pure mathematics: group theory, rings, fields and other abstract algebraic structures."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_anatomy",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_anatomy",
   "models_covered": 50,
   "name": "MMLU: Anatomy",
   "reasons": [],
   "summary": "MMLU subject subset: Human anatomical structures, organ systems and anatomical terminology, at a premedical or undergraduate level."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_astronomy",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_astronomy",
   "models_covered": 90,
   "name": "MMLU: Astronomy",
   "reasons": [],
   "summary": "MMLU subject subset: Celestial mechanics, stars, planets and cosmology, at an undergraduate astronomy-course level."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_biology",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_biology",
   "models_covered": 40,
   "name": "MMLU: Biology (subcategory)",
   "reasons": [],
   "summary": "The biology subcategory of MMLU: a rollup of the College Biology and High School Biology subjects, used by publishers that report MMLU at a coarser grain than all 57 subjects."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_business_ethics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_business_ethics",
   "models_covered": 90,
   "name": "MMLU: Business Ethics",
   "reasons": [],
   "summary": "MMLU subject subset: Ethical issues in corporate conduct, governance and stakeholder theory, as taught in a business-school ethics course."
  },
  {
   "aliases": [
    "MMLU CF",
    "Contamination-free MMLU"
   ],
   "canonical_id": "mmlu_cf",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_cf",
   "models_covered": 0,
   "name": "MMLU-CF",
   "reasons": [],
   "summary": "A 14-category, four-option knowledge test with a closed 10k test set, built so MMLU-style leakage is harder."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_chemistry",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_chemistry",
   "models_covered": 40,
   "name": "MMLU: Chemistry (subcategory)",
   "reasons": [],
   "summary": "The chemistry subcategory of MMLU: a rollup of the College Chemistry and High School Chemistry subjects, used by publishers that report MMLU at a coarser grain than all 57 subjects."
  },
  {
   "aliases": [
    "mmlu_cm_ck_vir",
    "Bridging-the-Gap MMLU-Clinical"
   ],
   "canonical_id": "mmlu_clinical_afr",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_clinical_afr",
   "models_covered": 0,
   "name": "MMLU clinical African languages (HELM)",
   "reasons": [],
   "summary": "HELM wrap of human-translated MMLU clinical knowledge, college medicine, and virology items in 11 African languages, scored by exact match."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_clinical_knowledge",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_clinical_knowledge",
   "models_covered": 90,
   "name": "MMLU: Clinical Knowledge",
   "reasons": [],
   "summary": "MMLU subject subset: General clinical medicine facts and patient-care knowledge, at a level below the licensing-exam-style Professional Medicine subject."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_college_biology",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_college_biology",
   "models_covered": 50,
   "name": "MMLU: College Biology",
   "reasons": [],
   "summary": "MMLU subject subset: Molecular and cellular biology, genetics and physiology, at the undergraduate level."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_college_chemistry",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_college_chemistry",
   "models_covered": 50,
   "name": "MMLU: College Chemistry",
   "reasons": [],
   "summary": "MMLU subject subset: General and organic chemistry at the undergraduate level."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_college_computer_science",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_college_computer_science",
   "models_covered": 50,
   "name": "MMLU: College Computer Science",
   "reasons": [],
   "summary": "MMLU subject subset: Algorithms, computability and computer science theory, at the undergraduate level."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_college_mathematics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_college_mathematics",
   "models_covered": 50,
   "name": "MMLU: College Mathematics",
   "reasons": [],
   "summary": "MMLU subject subset: Calculus, linear algebra and other undergraduate mathematics."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_college_medicine",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_college_medicine",
   "models_covered": 50,
   "name": "MMLU: College Medicine",
   "reasons": [],
   "summary": "MMLU subject subset: Medical-school coursework, distinct from the licensing-exam-style Professional Medicine subject."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_college_physics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_college_physics",
   "models_covered": 50,
   "name": "MMLU: College Physics",
   "reasons": [],
   "summary": "MMLU subject subset: Undergraduate mechanics, electromagnetism and thermodynamics, typically calculation-heavy."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_computer_science",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_computer_science",
   "models_covered": 40,
   "name": "MMLU: Computer Science (subcategory)",
   "reasons": [],
   "summary": "The computer science subcategory of MMLU: a rollup of four subjects, used by publishers that report MMLU at a coarser grain than all 57 subjects."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_computer_security",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_computer_security",
   "models_covered": 50,
   "name": "MMLU: Computer Security",
   "reasons": [],
   "summary": "MMLU subject subset: Cryptography, network security and systems-security concepts."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_conceptual_physics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_conceptual_physics",
   "models_covered": 50,
   "name": "MMLU: Conceptual Physics",
   "reasons": [],
   "summary": "MMLU subject subset: Qualitative, non-calculus physics concepts, as opposed to the calculation-heavy College Physics subject."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_econometrics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_econometrics",
   "models_covered": 50,
   "name": "MMLU: Econometrics",
   "reasons": [],
   "summary": "MMLU subject subset: Statistical methods applied to economic data: regression, estimation and hypothesis testing in an economics context."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_electrical_engineering",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_electrical_engineering",
   "models_covered": 50,
   "name": "MMLU: Electrical Engineering",
   "reasons": [],
   "summary": "MMLU subject subset: Circuits, signals and electronics fundamentals."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_elementary_mathematics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_elementary_mathematics",
   "models_covered": 50,
   "name": "MMLU: Elementary Mathematics",
   "reasons": [],
   "summary": "MMLU subject subset: Grade-school arithmetic and pre-algebra, including order-of-operations and word problems."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_formal_logic",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_formal_logic",
   "models_covered": 50,
   "name": "MMLU: Formal Logic",
   "reasons": [],
   "summary": "MMLU subject subset: Propositional and predicate logic, validity and inference rules, as taught in a philosophy department."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_global_facts",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_global_facts",
   "models_covered": 50,
   "name": "MMLU: Global Facts",
   "reasons": [],
   "summary": "MMLU subject subset: General-knowledge factual questions about world demographics, statistics and current facts, not tied to a single academic course."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_high_school_biology",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_biology",
   "models_covered": 50,
   "name": "MMLU: High School Biology",
   "reasons": [],
   "summary": "MMLU subject subset: Standard high-school biology curriculum: ecology, genetics and cell biology."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_high_school_chemistry",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_chemistry",
   "models_covered": 50,
   "name": "MMLU: High School Chemistry",
   "reasons": [],
   "summary": "MMLU subject subset: Standard high-school chemistry curriculum."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_high_school_computer_science",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_computer_science",
   "models_covered": 50,
   "name": "MMLU: High School Computer Science",
   "reasons": [],
   "summary": "MMLU subject subset: Introductory computer science and programming concepts at high-school level."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_high_school_european_history",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_european_history",
   "models_covered": 50,
   "name": "MMLU: High School European History",
   "reasons": [],
   "summary": "MMLU subject subset: European history from a standard high-school or AP-level course."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_high_school_geography",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_geography",
   "models_covered": 50,
   "name": "MMLU: High School Geography",
   "reasons": [],
   "summary": "MMLU subject subset: Physical and human geography at high-school level."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_high_school_government_and_politics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_government_and_politics",
   "models_covered": 50,
   "name": "MMLU: High School Government and Politics",
   "reasons": [],
   "summary": "MMLU subject subset: US government structure, civics and political institutions."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_high_school_macroeconomics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_macroeconomics",
   "models_covered": 50,
   "name": "MMLU: High School Macroeconomics",
   "reasons": [],
   "summary": "MMLU subject subset: Macroeconomic concepts -- GDP, inflation, fiscal and monetary policy -- at high-school/introductory level."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_high_school_mathematics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_mathematics",
   "models_covered": 50,
   "name": "MMLU: High School Mathematics",
   "reasons": [],
   "summary": "MMLU subject subset: Algebra, geometry and trigonometry at high-school level."
  },
  {
   "aliases": [
    "high_school_microeconomics"
   ],
   "canonical_id": "mmlu_high_school_microeconomics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_microeconomics",
   "models_covered": 50,
   "name": "MMLU: High School Microeconomics",
   "reasons": [],
   "summary": "Accuracy on MMLU's high school microeconomics questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "high_school_physics"
   ],
   "canonical_id": "mmlu_high_school_physics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_physics",
   "models_covered": 50,
   "name": "MMLU: High School Physics",
   "reasons": [],
   "summary": "Accuracy on MMLU's high school physics questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "high_school_psychology"
   ],
   "canonical_id": "mmlu_high_school_psychology",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_psychology",
   "models_covered": 50,
   "name": "MMLU: High School Psychology",
   "reasons": [],
   "summary": "Accuracy on MMLU's high school psychology questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "high_school_statistics"
   ],
   "canonical_id": "mmlu_high_school_statistics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_statistics",
   "models_covered": 50,
   "name": "MMLU: High School Statistics",
   "reasons": [],
   "summary": "Accuracy on MMLU's high school statistics questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "high_school_us_history"
   ],
   "canonical_id": "mmlu_high_school_us_history",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_us_history",
   "models_covered": 50,
   "name": "MMLU: High School US History",
   "reasons": [],
   "summary": "Accuracy on MMLU's high school us history questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "high_school_world_history"
   ],
   "canonical_id": "mmlu_high_school_world_history",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_high_school_world_history",
   "models_covered": 50,
   "name": "MMLU: High School World History",
   "reasons": [],
   "summary": "Accuracy on MMLU's high school world history questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "human_aging"
   ],
   "canonical_id": "mmlu_human_aging",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_human_aging",
   "models_covered": 50,
   "name": "MMLU: Human Aging",
   "reasons": [],
   "summary": "Accuracy on MMLU's human aging questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "human_sexuality"
   ],
   "canonical_id": "mmlu_human_sexuality",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_human_sexuality",
   "models_covered": 50,
   "name": "MMLU: Human Sexuality",
   "reasons": [],
   "summary": "Accuracy on MMLU's human sexuality questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "international_law"
   ],
   "canonical_id": "mmlu_international_law",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_international_law",
   "models_covered": 50,
   "name": "MMLU: International Law",
   "reasons": [],
   "summary": "Accuracy on MMLU's international law questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "jurisprudence"
   ],
   "canonical_id": "mmlu_jurisprudence",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_jurisprudence",
   "models_covered": 90,
   "name": "MMLU: Jurisprudence",
   "reasons": [],
   "summary": "Accuracy on MMLU's jurisprudence questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "logical_fallacies"
   ],
   "canonical_id": "mmlu_logical_fallacies",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_logical_fallacies",
   "models_covered": 50,
   "name": "MMLU: Logical Fallacies",
   "reasons": [],
   "summary": "Accuracy on MMLU's logical fallacies questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "machine_learning"
   ],
   "canonical_id": "mmlu_machine_learning",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_machine_learning",
   "models_covered": 50,
   "name": "MMLU: Machine Learning",
   "reasons": [],
   "summary": "Accuracy on MMLU's machine learning questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "management"
   ],
   "canonical_id": "mmlu_management",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_management",
   "models_covered": 50,
   "name": "MMLU: Management",
   "reasons": [],
   "summary": "Accuracy on MMLU's management questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "marketing"
   ],
   "canonical_id": "mmlu_marketing",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_marketing",
   "models_covered": 50,
   "name": "MMLU: Marketing",
   "reasons": [],
   "summary": "Accuracy on MMLU's marketing questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "medical_genetics"
   ],
   "canonical_id": "mmlu_medical_genetics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_medical_genetics",
   "models_covered": 50,
   "name": "MMLU: Medical Genetics",
   "reasons": [],
   "summary": "Accuracy on MMLU's medical genetics questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "miscellaneous"
   ],
   "canonical_id": "mmlu_miscellaneous",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_miscellaneous",
   "models_covered": 50,
   "name": "MMLU: Miscellaneous",
   "reasons": [],
   "summary": "Accuracy on MMLU's miscellaneous questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "moral_disputes"
   ],
   "canonical_id": "mmlu_moral_disputes",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_moral_disputes",
   "models_covered": 50,
   "name": "MMLU: Moral Disputes",
   "reasons": [],
   "summary": "Accuracy on MMLU's moral disputes questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "moral_scenarios"
   ],
   "canonical_id": "mmlu_moral_scenarios",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_moral_scenarios",
   "models_covered": 50,
   "name": "MMLU: Moral Scenarios",
   "reasons": [],
   "summary": "Accuracy on MMLU's moral scenarios questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "nutrition"
   ],
   "canonical_id": "mmlu_nutrition",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_nutrition",
   "models_covered": 50,
   "name": "MMLU: Nutrition",
   "reasons": [],
   "summary": "Accuracy on MMLU's nutrition questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "philosophy"
   ],
   "canonical_id": "mmlu_philosophy",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_philosophy",
   "models_covered": 50,
   "name": "MMLU: Philosophy",
   "reasons": [],
   "summary": "Accuracy on MMLU's philosophy questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "mmlu_pro_physics"
   ],
   "canonical_id": "mmlu_physics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_physics",
   "models_covered": 40,
   "name": "MMLU: Physics (unresolved key)",
   "reasons": [],
   "summary": "Disputed key on 40 frontier cards: possibly a mis-keyed MMLU-Pro Physics score, possibly a classic-MMLU STEM subcategory rollup. Neither reading is confirmed; pending a card re-key."
  },
  {
   "aliases": [
    "prehistory"
   ],
   "canonical_id": "mmlu_prehistory",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_prehistory",
   "models_covered": 50,
   "name": "MMLU: Prehistory",
   "reasons": [],
   "summary": "Accuracy on MMLU's prehistory questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_pro",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_pro",
   "models_covered": 354,
   "name": "MMLU-Pro",
   "reasons": [],
   "summary": "A harder, ten-option successor to MMLU with about 12,000 reasoning-heavy questions across 14 categories, built to restore headroom lost to MMLU's saturation at the frontier."
  },
  {
   "aliases": [
    "MMLU Pro Plus",
    "MMLU-Pro+"
   ],
   "canonical_id": "mmlu_pro_plus",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "mmlu_pro_plus",
   "models_covered": 0,
   "name": "MMLU-Pro+",
   "reasons": [],
   "summary": "A 12,032-question MMLU-Pro extension that adds multi-correct pairs to test higher-order reasoning and shortcut resistance."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_professional_accounting",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_professional_accounting",
   "models_covered": 90,
   "name": "MMLU: Professional Accounting",
   "reasons": [],
   "summary": "MMLU subject subset: Case-style problems in financial accounting, auditing, cost accounting, tax and business law."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_professional_law",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_professional_law",
   "models_covered": 90,
   "name": "MMLU: Professional Law",
   "reasons": [],
   "summary": "MMLU subject subset: Bar-exam-style fact patterns testing US law, and by far the largest of MMLU's 57 subjects."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_professional_medicine",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_professional_medicine",
   "models_covered": 50,
   "name": "MMLU: Professional Medicine",
   "reasons": [],
   "summary": "MMLU subject subset: USMLE-style clinical vignettes on diagnosis, mechanism and management across medical specialties."
  },
  {
   "aliases": [],
   "canonical_id": "mmlu_professional_psychology",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_professional_psychology",
   "models_covered": 50,
   "name": "MMLU: Professional Psychology",
   "reasons": [],
   "summary": "MMLU subject subset: Licensing-exam-style questions spanning clinical, developmental and social psychology, psychometrics and professional ethics."
  },
  {
   "aliases": [
    "MMLU-ProX: A Multilingual Benchmark for Advanced Large Language Model Evaluation"
   ],
   "canonical_id": "mmlu_prox",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_prox",
   "models_covered": 0,
   "name": "MMLU-ProX",
   "reasons": [],
   "summary": "A 29-language, expert-verified translation of MMLU-Pro with 11,829 identical questions per language, built to compare cross-linguistic reasoning rather than just English knowledge."
  },
  {
   "aliases": [
    "public_relations"
   ],
   "canonical_id": "mmlu_public_relations",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_public_relations",
   "models_covered": 50,
   "name": "MMLU: Public Relations",
   "reasons": [],
   "summary": "Accuracy on MMLU's public relations questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "MMLU Redux",
    "Are We Done with MMLU?"
   ],
   "canonical_id": "mmlu_redux",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_redux",
   "models_covered": 0,
   "name": "MMLU-Redux",
   "reasons": [],
   "summary": "A re-annotated slice of MMLU (5,700 items across 57 subjects in the 2.0 release) that tags label errors instead of adding new questions."
  },
  {
   "aliases": [
    "security_studies"
   ],
   "canonical_id": "mmlu_security_studies",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_security_studies",
   "models_covered": 50,
   "name": "MMLU: Security Studies",
   "reasons": [],
   "summary": "Accuracy on MMLU's security studies questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "sociology"
   ],
   "canonical_id": "mmlu_sociology",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_sociology",
   "models_covered": 50,
   "name": "MMLU: Sociology",
   "reasons": [],
   "summary": "Accuracy on MMLU's sociology questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "us_foreign_policy"
   ],
   "canonical_id": "mmlu_us_foreign_policy",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_us_foreign_policy",
   "models_covered": 50,
   "name": "MMLU: US Foreign Policy",
   "reasons": [],
   "summary": "Accuracy on MMLU's us foreign policy questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "virology"
   ],
   "canonical_id": "mmlu_virology",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_virology",
   "models_covered": 50,
   "name": "MMLU: Virology",
   "reasons": [],
   "summary": "Accuracy on MMLU's virology questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "world_religions"
   ],
   "canonical_id": "mmlu_world_religions",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmlu_world_religions",
   "models_covered": 50,
   "name": "MMLU: World Religions",
   "reasons": [],
   "summary": "Accuracy on MMLU's world religions questions, one of 57 subject tests of academic and professional knowledge."
  },
  {
   "aliases": [
    "Arabic MMLU (AceGPT)",
    "acegpt_MMLUArabic"
   ],
   "canonical_id": "mmluarabic",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmluarabic",
   "models_covered": 0,
   "name": "MMLUArabic (AceGPT translated MMLU)",
   "reasons": [],
   "summary": "AceGPT's GPT-3.5-Turbo translation of English MMLU into Arabic across 57 subjects; not the native-exam ArabicMMLU suite."
  },
  {
   "aliases": [
    "MMLU-SR",
    "MMLU Symbol Replacement",
    "MMLU-R"
   ],
   "canonical_id": "mmlusr",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "mmlusr",
   "models_covered": 0,
   "name": "MMLU-SR",
   "reasons": [],
   "summary": "An MMLU variant that replaces key terms with defined dummy symbols to separate conceptual reasoning from surface pattern matching."
  },
  {
   "aliases": [
    "MMMLU",
    "Multilingual MMLU"
   ],
   "canonical_id": "mmmlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmmlu",
   "models_covered": 2,
   "name": "Multilingual MMLU (MMMLU)",
   "reasons": [],
   "summary": "MMLU's test set professionally translated into 14 languages, testing whether a model's broad academic knowledge holds up outside English."
  },
  {
   "aliases": [
    "MMMLU Lite",
    "Global MMLU lite"
   ],
   "canonical_id": "mmmlu_lite",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "mmmlu_lite",
   "models_covered": 0,
   "name": "MMMLU-lite",
   "reasons": [],
   "summary": "A 19,950-item multilingual MMLU slice with 25 test questions per subject-language pair across 57 subjects and 14 languages."
  },
  {
   "aliases": [
    "Massive Multi-discipline Multimodal Understanding"
   ],
   "canonical_id": "mmmu",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "mmmu",
   "models_covered": 68,
   "name": "MMMU",
   "reasons": [],
   "summary": "11.5K college-level, image-paired exam questions across six disciplines, testing expert knowledge that requires reading a figure, chart or diagram."
  },
  {
   "aliases": [
    "MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark"
   ],
   "canonical_id": "mmmu_pro",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "mmmu_pro",
   "models_covered": 0,
   "name": "MMMU-Pro",
   "reasons": [],
   "summary": "A harder MMMU variant that filters out text-answerable questions, expands options to ten, and adds a vision-only setting where the question is embedded in a photo or screenshot."
  },
  {
   "aliases": [
    "BIG-bench mnist_ascii"
   ],
   "canonical_id": "mnist_ascii",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "mnist_ascii",
   "models_covered": 0,
   "name": "ASCII MNIST",
   "reasons": [],
   "summary": "A BIG-bench task that asks a model to identify MNIST digits rendered as ASCII art."
  },
  {
   "aliases": [
    "Discovering Language Model Behaviors with Model-Written Evaluations",
    "Anthropic Model-Written Evals",
    "MWE"
   ],
   "canonical_id": "model_written_evals",
   "category": "safety",
   "disposition": "unassessed",
   "id": "model_written_evals",
   "models_covered": 0,
   "name": "Model-Written Evaluations",
   "reasons": [],
   "summary": "154 model-generated, human-filtered yes/no datasets probing a model's persona, sycophancy and advanced-AI-risk tendencies rather than testing right-or-wrong knowledge."
  },
  {
   "aliases": [
    "BIG-bench modified_arithmetic"
   ],
   "canonical_id": "modified_arithmetic",
   "category": "math",
   "disposition": "unassessed",
   "id": "modified_arithmetic",
   "models_covered": 0,
   "name": "Modified Arithmetic",
   "reasons": [],
   "summary": "A BIG-bench task testing whether a model learns arithmetic operations followed by an unusual +1 rule from examples."
  },
  {
   "aliases": [
    "Molecular IQ",
    "OpenCompass molculariq"
   ],
   "canonical_id": "molculariq",
   "category": "domain",
   "disposition": "unassessed",
   "id": "molculariq",
   "models_covered": 0,
   "name": "MolecularIQ",
   "reasons": [],
   "summary": "An OpenCompass wrapper for MolecularIQ, which evaluates chemical reasoning over molecular structures with symbolic verification."
  },
  {
   "aliases": [
    "MolInstructions_chem",
    "Mol-Instructions",
    "OpenCompass MolInstructions chem"
   ],
   "canonical_id": "molinstructions_chem",
   "category": "domain",
   "disposition": "unassessed",
   "id": "molinstructions_chem",
   "models_covered": 0,
   "name": "Mol-Instructions molecule-oriented tasks",
   "reasons": [],
   "summary": "An OpenCompass wrapper for six molecule-oriented Mol-Instructions tasks covering molecular descriptions, design, reactions, properties, and retrosynthesis."
  },
  {
   "aliases": [
    "Judging Moral Permissibility"
   ],
   "canonical_id": "moral_permissibility",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "moral_permissibility",
   "models_covered": 0,
   "name": "Moral Permissibility",
   "reasons": [],
   "summary": "A 342-item BIG-bench task that asks yes or no whether an action in a short moral-dilemma story is permissible, using majority human labels."
  },
  {
   "aliases": [
    "MoralStories"
   ],
   "canonical_id": "moral_stories",
   "category": "safety",
   "disposition": "unassessed",
   "id": "moral_stories",
   "models_covered": 0,
   "name": "Moral Stories",
   "reasons": [],
   "summary": "lm-eval ranks a crowd-written moral action against an immoral one, given a social norm, situation and intention from the 12k-story Moral Stories corpus."
  },
  {
   "aliases": [
    "Moral Reasoning under Uncertainty",
    "inspect_evals/moru",
    "moru-benchmark"
   ],
   "canonical_id": "moru",
   "category": "safety",
   "disposition": "unassessed",
   "id": "moru",
   "models_covered": 0,
   "name": "MORU (Moral Reasoning under Uncertainty)",
   "reasons": [],
   "summary": "Inspect Evals MORU: 201 multilingual moral-uncertainty scenarios scored by LLM graders on 16 binary ethical dimensions."
  },
  {
   "aliases": [
    "Movie Dialogue",
    "movie_dialog"
   ],
   "canonical_id": "movie_dialog_same_or_different",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "movie_dialog_same_or_different",
   "models_covered": 0,
   "name": "Movie Dialogue Same or Different (BIG-bench)",
   "reasons": [],
   "summary": "A 50,000-item BIG-bench binary task: decide whether two adjacent movie-script sentences were spoken by the same person."
  },
  {
   "aliases": [
    "BIG-bench movie_recommendation"
   ],
   "canonical_id": "movie_recommendation",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "movie_recommendation",
   "models_covered": 0,
   "name": "Movie Recommendation (BIG-bench)",
   "reasons": [],
   "summary": "A 500-item BIG-bench four-way quiz: pick a further liked movie given four titles a MovieLens user already liked."
  },
  {
   "aliases": [
    "MP-20",
    "MP20",
    "opencompass/mp20"
   ],
   "canonical_id": "mp20",
   "category": "domain",
   "disposition": "unassessed",
   "id": "mp20",
   "models_covered": 0,
   "name": "MP-20 (OpenCompass)",
   "reasons": [],
   "summary": "OpenCompass LLM wrap of MP-20: emit lattice and atomic sites as JSON, scored by StructureMatcher match rate and RMS."
  },
  {
   "aliases": [],
   "canonical_id": "mrag",
   "category": "domain",
   "disposition": "unassessed",
   "id": "mrag",
   "models_covered": 0,
   "name": "MRAG",
   "reasons": [],
   "summary": "MRAG evaluates retrieval-augmented generation for biomedical question answering in English and Chinese using Wikipedia and PubMed corpora."
  },
  {
   "aliases": [],
   "canonical_id": "mrag_bench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "mrag_bench",
   "models_covered": 0,
   "name": "MRAG-Bench",
   "reasons": [],
   "summary": "MRAG-Bench evaluates whether vision-language models can retrieve and use visual knowledge for multimodal question answering."
  },
  {
   "aliases": [
    "MS MARCO",
    "MSMARCO",
    "Microsoft MAchine Reading COmprehension",
    "msmarco_regular",
    "msmarco_trec"
   ],
   "canonical_id": "msmarco",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "msmarco",
   "models_covered": 0,
   "name": "MS MARCO (HELM passage ranking)",
   "reasons": [],
   "summary": "HELM's MS MARCO passage-ranking wrap: an LLM says Yes or No whether a passage answers a Bing query, then HELM ranks those decisions."
  },
  {
   "aliases": [
    "MT-bench"
   ],
   "canonical_id": "mt_bench",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "mt_bench",
   "models_covered": 46,
   "name": "MT-Bench",
   "reasons": [],
   "summary": "Eighty multi-turn chat questions graded by an LLM judge as a fast, repeatable stand-in for human conversational preference."
  },
  {
   "aliases": [
    "Massive Text Embedding Benchmark",
    "MMTEB"
   ],
   "canonical_id": "mteb",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "mteb",
   "models_covered": 0,
   "name": "MTEB (Massive Text Embedding Benchmark)",
   "reasons": [],
   "summary": "A multi-task suite that scores text embedding models on retrieval, classification, clustering, reranking, similarity, summarization and pair classification."
  },
  {
   "aliases": [
    "MTEB-BR: A Multilingual Benchmark for Issue Resolving"
   ],
   "canonical_id": "mteb_br",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "mteb_br",
   "models_covered": 0,
   "name": "MTEB-BR",
   "reasons": [],
   "summary": "MTEB-BR evaluates agents that modify repositories to resolve issues across Java, TypeScript, JavaScript, Go, Rust, C and C++."
  },
  {
   "aliases": [],
   "canonical_id": "mteb_classification",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "mteb_classification",
   "models_covered": 94,
   "name": "MTEB Classification",
   "reasons": [],
   "summary": "Trains a logistic regression probe on a model's embeddings and scores accuracy on 12 classification datasets in English and other languages."
  },
  {
   "aliases": [],
   "canonical_id": "mteb_clustering",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "mteb_clustering",
   "models_covered": 94,
   "name": "MTEB Clustering",
   "reasons": [],
   "summary": "Runs mini-batch k-means over a model's embeddings and scores the clusters against ground-truth labels with V-measure, across 11 mostly-English datasets."
  },
  {
   "aliases": [
    "MTEB score",
    "MTEB Task Mean",
    "MTEB Task Type Mean"
   ],
   "canonical_id": "mteb_overall",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "mteb_overall",
   "models_covered": 96,
   "name": "MTEB Overall (leaderboard average)",
   "reasons": [],
   "summary": "The blended average the public MTEB leaderboard shows across a model's task-type scores; the number most people mean when they say 'MTEB score'."
  },
  {
   "aliases": [],
   "canonical_id": "mteb_pair_classification",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "mteb_pair_classification",
   "models_covered": 93,
   "name": "MTEB Pair Classification",
   "reasons": [],
   "summary": "Labels sentence pairs as duplicates or not from cosine similarity and scores the ranking by average precision, across 3 primarily-English MTEB datasets."
  },
  {
   "aliases": [],
   "canonical_id": "mteb_reranking",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "mteb_reranking",
   "models_covered": 93,
   "name": "MTEB Reranking",
   "reasons": [],
   "summary": "Reorders a fixed candidate list of passages for a query by embedding similarity and scores the reordering by MAP, across 4 mostly-English datasets."
  },
  {
   "aliases": [],
   "canonical_id": "mteb_retrieval",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "mteb_retrieval",
   "models_covered": 96,
   "name": "MTEB Retrieval",
   "reasons": [],
   "summary": "Ranks passages in a corpus by relevance to a query using embedding similarity, scored by nDCG@10 across 15 mostly-English MTEB datasets."
  },
  {
   "aliases": [
    "MTEB Semantic Textual Similarity"
   ],
   "canonical_id": "mteb_sts",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "mteb_sts",
   "models_covered": 93,
   "name": "MTEB STS (Semantic Textual Similarity)",
   "reasons": [],
   "summary": "Correlates embedding cosine similarity with human-rated sentence-pair similarity scores, across 10 datasets spanning up to 18 languages."
  },
  {
   "aliases": [],
   "canonical_id": "mteb_summarization",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "mteb_summarization",
   "models_covered": 93,
   "name": "MTEB Summarization",
   "reasons": [],
   "summary": "Scores whether embedding similarity to human-written summaries predicts human quality ratings of machine summaries, on a single English dataset."
  },
  {
   "aliases": [],
   "canonical_id": "mtr_bench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "mtr_bench",
   "models_covered": 0,
   "name": "MTR-Bench",
   "reasons": [],
   "summary": "MTR-Bench evaluates multi-turn interactive reasoning across 40 tasks and 3,600 instances in four task classes."
  },
  {
   "aliases": [],
   "canonical_id": "mtr_suite",
   "category": "composite",
   "disposition": "unassessed",
   "id": "mtr_suite",
   "models_covered": 0,
   "name": "MTR-Suite",
   "reasons": [],
   "summary": "MTR-Suite audits and synthesizes conversational retrieval benchmarks and introduces a production-style MTR-Bench."
  },
  {
   "aliases": [
    "MTS-Dialog",
    "MTS_Dialogue-Clinical_Note",
    "mts_dialog_perplexity"
   ],
   "canonical_id": "mts_dialog",
   "category": "domain",
   "disposition": "unassessed",
   "id": "mts_dialog",
   "models_covered": 0,
   "name": "MTS-Dialog (lm-eval)",
   "reasons": [],
   "summary": "lm-eval wrap of MTS-Dialog: write an English clinical-note section from a short doctor-patient dialogue, scored with overlap metrics."
  },
  {
   "aliases": [
    "MTSamples-Procedures",
    "mtsample_procedure"
   ],
   "canonical_id": "mtsamples_procedures",
   "category": "domain",
   "disposition": "unassessed",
   "id": "mtsamples_procedures",
   "models_covered": 0,
   "name": "MTSamples Procedures (MedHELM)",
   "reasons": [],
   "summary": "MedHELM wrap of MTSamples surgical notes: generate a plan or findings from an operative transcription, scored by an LLM jury plus overlap metrics."
  },
  {
   "aliases": [
    "MTSamples",
    "mtsamples_processed"
   ],
   "canonical_id": "mtsamples_replicate",
   "category": "domain",
   "disposition": "unassessed",
   "id": "mtsamples_replicate",
   "models_covered": 0,
   "name": "MTSamples Replicate (MedHELM)",
   "reasons": [],
   "summary": "MedHELM wrap of mixed-specialty MTSamples notes: generate a treatment plan from a clinical transcription, scored by an LLM jury plus overlap metrics."
  },
  {
   "aliases": [
    "mult_data_wrangling",
    "BIG-bench Data Wrangling"
   ],
   "canonical_id": "mult_data_wrangling",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "mult_data_wrangling",
   "models_covered": 0,
   "name": "Data Wrangling (BIG-bench)",
   "reasons": [],
   "summary": "A 246-file BIG-bench suite of string wrangling problems: infer a format change from few-shot pairs and emit the transformed string."
  },
  {
   "aliases": [],
   "canonical_id": "multi",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "multi",
   "models_covered": 0,
   "name": "MULTI",
   "reasons": [],
   "summary": "MULTI evaluates Chinese multimodal understanding with more than 18,000 authentic examination questions and hard and in-context variants."
  },
  {
   "aliases": [],
   "canonical_id": "multi_bench",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "multi_bench",
   "models_covered": 0,
   "name": "MULTI-Bench",
   "reasons": [],
   "summary": "MULTI-Bench evaluates spoken dialogue models on multi-turn emotional intelligence through basic understanding and advanced support tracks."
  },
  {
   "aliases": [
    "Multi-SWE-bench: A Multilingual Benchmark for Issue Resolving"
   ],
   "canonical_id": "multi_swe_bench",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multi_swe_bench",
   "models_covered": 0,
   "name": "Multi-SWE-bench",
   "reasons": [],
   "summary": "Multi-SWE-bench evaluates agents that modify repositories to resolve issues across Java, TypeScript, JavaScript, Go, Rust, C and C++."
  },
  {
   "aliases": [
    "MultiBLiMP",
    "multiblimp",
    "MultiBLiMP 1.0"
   ],
   "canonical_id": "multiblimp",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "multiblimp",
   "models_covered": 0,
   "name": "MultiBLiMP 1.0",
   "reasons": [],
   "summary": "101-language minimal-pair benchmark of subject-verb agreement, scored by whether a model assigns higher probability to the grammatical sentence."
  },
  {
   "aliases": [
    "MultiEmo",
    "multiemo"
   ],
   "canonical_id": "multiemo",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "multiemo",
   "models_covered": 0,
   "name": "MultiEmo (BIG-bench)",
   "reasons": [],
   "summary": "A BIG-bench wrap of MultiEmo: four-way sentiment on consumer reviews in 11 languages at document and sentence level."
  },
  {
   "aliases": [
    "MultiIF",
    "facebook/Multi-IF"
   ],
   "canonical_id": "multiif",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "multiif",
   "models_covered": 0,
   "name": "Multi-IF",
   "reasons": [],
   "summary": "IFEval-style verifiable instruction following extended to three-turn conversations in eight languages, 4,501 dialogues."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e",
   "models_covered": 37,
   "name": "MultiPL-E",
   "reasons": [],
   "summary": "MultiPL-E mechanically translates the HumanEval and MBPP Python code-generation benchmarks into 18+ other programming languages and scores pass@k in each."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_cpp",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_cpp",
   "models_covered": 37,
   "name": "MultiPL-E: C++",
   "reasons": [],
   "summary": "The C++ subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into C++ and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_csharp",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_csharp",
   "models_covered": 135,
   "name": "MultiPL-E: C#",
   "reasons": [],
   "summary": "The C# subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into C# and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_go",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_go",
   "models_covered": 37,
   "name": "MultiPL-E: Go",
   "reasons": [],
   "summary": "The Go subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into Go and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_java",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_java",
   "models_covered": 37,
   "name": "MultiPL-E: Java",
   "reasons": [],
   "summary": "The Java subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into Java and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_javascript",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_javascript",
   "models_covered": 37,
   "name": "MultiPL-E: JavaScript",
   "reasons": [],
   "summary": "The JavaScript subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into JavaScript and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_julia",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_julia",
   "models_covered": 129,
   "name": "MultiPL-E: Julia",
   "reasons": [],
   "summary": "The Julia subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into Julia and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_kotlin",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_kotlin",
   "models_covered": 135,
   "name": "MultiPL-E: Kotlin (unconfirmed)",
   "reasons": [],
   "summary": "A Kotlin subset is not present in the MultiPL-E paper, repository or dataset card checked for this page; what this key measures is not established."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_lua",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_lua",
   "models_covered": 135,
   "name": "MultiPL-E: Lua",
   "reasons": [],
   "summary": "The Lua subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into Lua and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_perl",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_perl",
   "models_covered": 129,
   "name": "MultiPL-E: Perl",
   "reasons": [],
   "summary": "The Perl subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into Perl and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_php",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_php",
   "models_covered": 135,
   "name": "MultiPL-E: PHP",
   "reasons": [],
   "summary": "The PHP subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into PHP and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_python",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_python",
   "models_covered": 37,
   "name": "MultiPL-E: Python",
   "reasons": [],
   "summary": "The Python subset of MultiPL-E: the original, untranslated HumanEval and MBPP problems, used as the harness's reference language."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_r",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_r",
   "models_covered": 129,
   "name": "MultiPL-E: R",
   "reasons": [],
   "summary": "The R subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into R and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_ruby",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_ruby",
   "models_covered": 135,
   "name": "MultiPL-E: Ruby",
   "reasons": [],
   "summary": "The Ruby subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into Ruby and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_rust",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_rust",
   "models_covered": 37,
   "name": "MultiPL-E: Rust",
   "reasons": [],
   "summary": "The Rust subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into Rust and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_scala",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_scala",
   "models_covered": 135,
   "name": "MultiPL-E: Scala",
   "reasons": [],
   "summary": "The Scala subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into Scala and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_swift",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_swift",
   "models_covered": 135,
   "name": "MultiPL-E: Swift",
   "reasons": [],
   "summary": "The Swift subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into Swift and scored with pass@1."
  },
  {
   "aliases": [],
   "canonical_id": "multipl_e_typescript",
   "category": "coding",
   "disposition": "unassessed",
   "id": "multipl_e_typescript",
   "models_covered": 37,
   "name": "MultiPL-E: TypeScript",
   "reasons": [],
   "summary": "The TypeScript subset of MultiPL-E: HumanEval and MBPP function-completion problems translated into TypeScript and scored with pass@1."
  },
  {
   "aliases": [
    "multistep_arithmetic",
    "Multi-Step Arithmetic Two",
    "multistep_arithmetic_two"
   ],
   "canonical_id": "multistep_arithmetic",
   "category": "math",
   "disposition": "unassessed",
   "id": "multistep_arithmetic",
   "models_covered": 0,
   "name": "Multistep Arithmetic (BIG-bench)",
   "reasons": [],
   "summary": "A programmatic BIG-bench generator of nested integer sums, differences and products; BBH freezes 250 two-level expressions as Multistep Arithmetic Two."
  },
  {
   "aliases": [
    "muslim_violence_bias",
    "Muslim-Violence Bias"
   ],
   "canonical_id": "muslim_violence_bias",
   "category": "safety",
   "disposition": "unassessed",
   "id": "muslim_violence_bias",
   "models_covered": 0,
   "name": "Muslim-Violence Bias (BIG-bench)",
   "reasons": [],
   "summary": "A programmatic BIG-bench probe that compares violent completions after Muslim prompts versus matched Christian prompts, scoring bias on [\u22121, 0]."
  },
  {
   "aliases": [
    "Multistep Soft Reasoning"
   ],
   "canonical_id": "musr",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "musr",
   "models_covered": 221,
   "name": "MuSR",
   "reasons": [],
   "summary": "Algorithmically generated murder mysteries, object placement puzzles and team allocation problems needing long-range narrative reasoning."
  },
  {
   "aliases": [
    "MuTual"
   ],
   "canonical_id": "mutual",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "mutual",
   "models_covered": 0,
   "name": "MuTual",
   "reasons": [],
   "summary": "8,860 four-way response-selection dialogues rewritten from Chinese high-school English listening tests, scored with R@1, R@2 and MRR."
  },
  {
   "aliases": [],
   "canonical_id": "mv_bench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "mv_bench",
   "models_covered": 0,
   "name": "MV-Bench",
   "reasons": [],
   "summary": "MV-Bench evaluates multimodal models that generate coordinated multi-view visualization interfaces from visual and data specifications."
  },
  {
   "aliases": [],
   "canonical_id": "mv_dvrk",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "mv_dvrk",
   "models_covered": 0,
   "name": "MV-dVRK",
   "reasons": [],
   "summary": "MV-dVRK evaluates multi-view 3D reconstruction methods on synchronized stereo endoscopic views with surgical geometry and camera poses."
  },
  {
   "aliases": [
    "N2C2-CT",
    "n2c2 2018 Track 1",
    "n2c2 cohort selection"
   ],
   "canonical_id": "n2c2_ct_matching",
   "category": "domain",
   "disposition": "unassessed",
   "id": "n2c2_ct_matching",
   "models_covered": 0,
   "name": "N2C2-CT Matching (HELM)",
   "reasons": [],
   "summary": "MedHELM yes/no matching of n2c2 2018 patients to one inclusion criterion from de-identified notes, scored with exact match."
  },
  {
   "aliases": [
    "The NarrativeQA Reading Comprehension Challenge"
   ],
   "canonical_id": "narrativeqa",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "narrativeqa",
   "models_covered": 0,
   "name": "NarrativeQA",
   "reasons": [],
   "summary": "Free-form questions about entire books and movie scripts, scored against human reference answers, though most harnesses answer from a summary rather than the full narrative."
  },
  {
   "aliases": [
    "Natural Instructions",
    "Natural-Instructions",
    "NI v1"
   ],
   "canonical_id": "natural_instructions",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "natural_instructions",
   "models_covered": 0,
   "name": "natural_instructions (BIG-bench Natural Instructions)",
   "reasons": [],
   "summary": "BIG-bench packaging of Natural Instructions v1: 61 English generation tasks with crowdsourcing instructions, 193,250 items, scored with ROUGE-Lsum."
  },
  {
   "aliases": [
    "NaturalQA",
    "NaturalQuestions",
    "natural_qa_closedbook",
    "natural_qa_openbook_longans",
    "natural_qa_openbook_wiki"
   ],
   "canonical_id": "natural_qa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "natural_qa",
   "models_covered": 0,
   "name": "Natural Questions (HELM)",
   "reasons": [],
   "summary": "HELM's Natural Questions wrap: short answers to real Google searches, in closed-book, long-answer-context, or full-Wikipedia-page modes."
  },
  {
   "aliases": [
    "Navigation",
    "BBH navigate"
   ],
   "canonical_id": "navigate",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "navigate",
   "models_covered": 0,
   "name": "Navigate (BIG-bench)",
   "reasons": [],
   "summary": "A 1,000-item BIG-bench True/False task: after synthetic walk instructions, say whether the agent is back at the start."
  },
  {
   "aliases": [
    "NeedleBench v1",
    "needlebench"
   ],
   "canonical_id": "needlebench",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "needlebench",
   "models_covered": 0,
   "name": "NeedleBench",
   "reasons": [],
   "summary": "OpenCompass long-context suite that plants synthetic needles in English and Chinese haystacks at chosen lengths and depths, plus an Ancestral Trace Challenge."
  },
  {
   "aliases": [
    "needlebench_v2",
    "NeedleBench v2"
   ],
   "canonical_id": "needlebench_v2",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "needlebench_v2",
   "models_covered": 0,
   "name": "NeedleBench V2",
   "reasons": [],
   "summary": "OpenCompass revision of NeedleBench: equal-weight retrieval scores, fictional multi-needle facts, and power-of-two Ancestral Trace Challenge depths."
  },
  {
   "aliases": [
    "Nejmaibench",
    "nephSAP"
   ],
   "canonical_id": "nejm_ai_benchmark",
   "category": "domain",
   "disposition": "unassessed",
   "id": "nejm_ai_benchmark",
   "models_covered": 0,
   "name": "NEJMAI / nephSAP nephrology benchmark",
   "reasons": [],
   "summary": "OpenCompass's NEJMAI wrapper evaluates zero-shot multiple-choice answering on 858 nephSAP nephrology questions."
  },
  {
   "aliases": [],
   "canonical_id": "newsqa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "newsqa",
   "models_covered": 0,
   "name": "NewsQA",
   "reasons": [],
   "summary": "NewsQA tests whether a model can answer questions with text spans from CNN news articles, including unanswerable questions."
  },
  {
   "aliases": [
    "inspect_evals/niah",
    "Needle in a Haystack (Inspect Evals)"
   ],
   "canonical_id": "niah",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "niah",
   "models_covered": 0,
   "name": "NIAH (Inspect Evals Needle in a Haystack)",
   "reasons": [],
   "summary": "Inspect Evals NIAH: plant English needles in long haystacks and score recall with a 1\u201310 LLM judge across a length-by-depth grid."
  },
  {
   "aliases": [
    "Understanding Grammar of Unseen Words"
   ],
   "canonical_id": "nonsense_words_grammar",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "nonsense_words_grammar",
   "models_covered": 0,
   "name": "Nonsense Words Grammar (BIG-bench)",
   "reasons": [],
   "summary": "A 50-item BIG-bench multiple-choice task: infer part of speech or role for English-like nonsense words from context."
  },
  {
   "aliases": [],
   "canonical_id": "noreval",
   "category": "composite",
   "disposition": "unassessed",
   "id": "noreval",
   "models_covered": 0,
   "name": "NorEval",
   "reasons": [],
   "summary": "NorEval is a 24-dataset Norwegian language evaluation benchmark integrated into lm-evaluation-harness."
  },
  {
   "aliases": [
    "BIG-bench novel_concepts"
   ],
   "canonical_id": "novel_concepts",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "novel_concepts",
   "models_covered": 0,
   "name": "Novel Concepts (BIG-bench)",
   "reasons": [],
   "summary": "A 32-item BIG-bench Lite multiple-choice task: name the concept that unites two to four unlike English entities."
  },
  {
   "aliases": [
    "novelty-bench",
    "NoveltyBench: Evaluating Language Models for Humanlike Diversity"
   ],
   "canonical_id": "novelty_bench",
   "category": "generation",
   "disposition": "unassessed",
   "id": "novelty_bench",
   "models_covered": 0,
   "name": "NoveltyBench",
   "reasons": [],
   "summary": "Tests whether a model can produce several genuinely different, still-good answers to the same prompt across repeated samples, rather than near-duplicate outputs; current models fall well short of human writers."
  },
  {
   "aliases": [],
   "canonical_id": "nphardeval",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "nphardeval",
   "models_covered": 0,
   "name": "NPHardEval",
   "reasons": [],
   "summary": "NPHardEval tests large language models on 900 algorithmic questions spanning complexity classes through NP-hard problems."
  },
  {
   "aliases": [
    "nqcn",
    "NaturalQuestionDatasetCN"
   ],
   "canonical_id": "nq_cn",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "nq_cn",
   "models_covered": 0,
   "name": "NQ-CN (OpenCompass)",
   "reasons": [],
   "summary": "OpenCompass Chinese-prompt Natural Questions wrap: zero-shot short-answer generation scored by exact match on local jsonl files."
  },
  {
   "aliases": [
    "NQ-open",
    "nq-open",
    "Natural Questions Open"
   ],
   "canonical_id": "nq_open",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "nq_open",
   "models_covered": 0,
   "name": "NQ-Open",
   "reasons": [],
   "summary": "Closed-book short-answer QA on real Google queries from Natural Questions, usually scored by exact match on the public 3,610-item original NQ-Open dev split."
  },
  {
   "aliases": [],
   "canonical_id": "nyu_llm_ctf_ctftiny",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "nyu_llm_ctf_ctftiny",
   "models_covered": 0,
   "name": "NYU-LLM-CTF/CTFTiny",
   "reasons": [],
   "summary": "NYU-LLM-CTF/CTFTiny is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "nyu_llm_ctf_nyu_ctf_bench",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "nyu_llm_ctf_nyu_ctf_bench",
   "models_covered": 0,
   "name": "NYU-LLM-CTF/NYU_CTF_Bench",
   "reasons": [],
   "summary": "NYU-LLM-CTF/NYU_CTF_Bench is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [
    "Brazilian Bar Exam"
   ],
   "canonical_id": "oab_exams",
   "category": "domain",
   "disposition": "unassessed",
   "id": "oab_exams",
   "models_covered": 0,
   "name": "OAB Exams",
   "reasons": [],
   "summary": "OAB Exams evaluates Portuguese legal question answering on Brazilian bar examinations from 2010 through 2018."
  },
  {
   "aliases": [
    "BBH object_counting"
   ],
   "canonical_id": "object_counting",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "object_counting",
   "models_covered": 0,
   "name": "Object Counting (BIG-bench)",
   "reasons": [],
   "summary": "A 1,000-item BIG-bench free-response task: count one object class in a short English possession list, ignoring distractors."
  },
  {
   "aliases": [
    "OBQA"
   ],
   "canonical_id": "openbookqa",
   "category": "knowledge",
   "disposition": "alias",
   "id": "obqa",
   "models_covered": 0,
   "name": "OpenBookQA",
   "reasons": [
    "independently reviewed alias; retain canonical benchmark and protocol distinctions"
   ],
   "summary": "OpenBookQA tests multi-step science question answering with a small open book of facts and four answer choices."
  },
  {
   "aliases": [],
   "canonical_id": "ocrbench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "ocrbench",
   "models_covered": 21,
   "name": "OCRBench",
   "reasons": [],
   "summary": "1,000 hand-verified questions across five OCR task types, testing whether a multimodal model can read and reason about text embedded in images."
  },
  {
   "aliases": [
    "BIG-bench odd_one_out"
   ],
   "canonical_id": "odd_one_out",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "odd_one_out",
   "models_covered": 0,
   "name": "Odd One Out (BIG-bench)",
   "reasons": [],
   "summary": "An 86-item BIG-bench multiple-choice task: pick the word that does not belong in a four-to-six-word English list."
  },
  {
   "aliases": [],
   "canonical_id": "oecd_integrity",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "oecd_integrity",
   "models_covered": 0,
   "name": "OECD Integrity",
   "reasons": [],
   "summary": "OECD Integrity is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "oecd_outlook",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "oecd_outlook",
   "models_covered": 0,
   "name": "OECD Outlook",
   "reasons": [],
   "summary": "OECD Outlook is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "oecd_pisa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "oecd_pisa",
   "models_covered": 0,
   "name": "OECD PISA",
   "reasons": [],
   "summary": "OECD PISA is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "oecd_statistics",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "oecd_statistics",
   "models_covered": 0,
   "name": "OECD Statistics",
   "reasons": [],
   "summary": "OECD Statistics is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "ojbench",
   "category": "coding",
   "disposition": "unassessed",
   "id": "ojbench",
   "models_covered": 0,
   "name": "OJBench",
   "reasons": [],
   "summary": "OJBench evaluates competitive-level code reasoning on 232 NOI and ICPC programming problems."
  },
  {
   "aliases": [
    "arc_multilingual",
    "okapi/arc_multilingual",
    "m_arc"
   ],
   "canonical_id": "okapi_arc_multilingual",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "okapi_arc_multilingual",
   "models_covered": 0,
   "name": "Okapi ARC multilingual (lm-eval)",
   "reasons": [],
   "summary": "lm-eval group of GPT-3.5-translated ARC-Challenge questions in 31 languages on alexandrainst/m_arc, scored with acc and acc_norm."
  },
  {
   "aliases": [
    "hellaswag_multilingual",
    "m_hellaswag",
    "alexandrainst/m_hellaswag"
   ],
   "canonical_id": "okapi_hellaswag_multilingual",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "okapi_hellaswag_multilingual",
   "models_covered": 0,
   "name": "Okapi multilingual HellaSwag",
   "reasons": [],
   "summary": "lm-eval group of GPT-translated HellaSwag val sets in 30 languages; four-way continuation, scored acc and acc_norm."
  },
  {
   "aliases": [
    "m_mmlu",
    "mmlu_multilingual",
    "alexandrainst/m_mmlu"
   ],
   "canonical_id": "okapi_mmlu_multilingual",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "okapi_mmlu_multilingual",
   "models_covered": 0,
   "name": "Okapi multilingual MMLU",
   "reasons": [],
   "summary": "lm-eval tag m_mmlu: ChatGPT-translated MMLU in 34 language configs, four-choice accuracy on the test split."
  },
  {
   "aliases": [
    "truthfulqa_multilingual",
    "m_truthfulqa",
    "alexandrainst/m_truthfulqa"
   ],
   "canonical_id": "okapi_truthfulqa_multilingual",
   "category": "safety",
   "disposition": "unassessed",
   "id": "okapi_truthfulqa_multilingual",
   "models_covered": 0,
   "name": "Okapi multilingual TruthfulQA",
   "reasons": [],
   "summary": "lm-eval group of GPT-translated TruthfulQA val items in 31 languages, scored as MC1 accuracy and MC2 probability mass."
  },
  {
   "aliases": [
    "MedLFQA",
    "dmis-lab/MedLFQA",
    "olaph_perplexity"
   ],
   "canonical_id": "olaph",
   "category": "domain",
   "disposition": "unassessed",
   "id": "olaph",
   "models_covered": 0,
   "name": "OLAPH / MedLFQA",
   "reasons": [],
   "summary": "lm-eval wrap of MedLFQA: English biomedical long answers scored with BLEU, ROUGE, BERTScore and BLEURT on a 10% slice."
  },
  {
   "aliases": [
    "OlymMATH",
    "OlymMATH-EN",
    "OlymMATH-ZH",
    "OlymMATH-HARD",
    "OlymMATH-EASY"
   ],
   "canonical_id": "olymmath",
   "category": "math",
   "disposition": "unassessed",
   "id": "olymmath",
   "models_covered": 0,
   "name": "OlymMATH",
   "reasons": [],
   "summary": "A 350-problem bilingual Olympiad math suite: 200 numeric easy/hard items plus 150 Lean 4 proofs, after AIME and MATH stopped separating frontier models."
  },
  {
   "aliases": [
    "OlympiadBench",
    "Olympiad Bench"
   ],
   "canonical_id": "olympiadbench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "olympiadbench",
   "models_covered": 0,
   "name": "OlympiadBench",
   "reasons": [],
   "summary": "8,476 olympiad-level bilingual math and physics problems, many with figures, from contests and the Chinese gaokao; GPT-4V scored 17.97% on the full set at release."
  },
  {
   "aliases": [
    "OmniMATH",
    "Omni-Math"
   ],
   "canonical_id": "omni_math",
   "category": "math",
   "disposition": "unassessed",
   "id": "omni_math",
   "models_covered": 0,
   "name": "Omni-MATH",
   "reasons": [],
   "summary": "A 4,428-problem olympiad-level mathematics benchmark built after GSM8K and MATH became easy for frontier models, graded by an LLM judge rather than exact string match."
  },
  {
   "aliases": [
    "onet_m6",
    "O-NET M6",
    "thai-onet-m6-exam",
    "Ordinary National Educational Test"
   ],
   "canonical_id": "onet",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "onet",
   "models_covered": 0,
   "name": "O-NET (Inspect Evals)",
   "reasons": [],
   "summary": "Inspect Evals wrap of Thai O-NET M6: filtered multiple-choice items in Thai and English across five school subjects."
  },
  {
   "aliases": [
    "OASST1",
    "OpenAssistant Conversations",
    "OpenAssistant/oasst1"
   ],
   "canonical_id": "open_assistant",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "open_assistant",
   "models_covered": 0,
   "name": "Open Assistant (HELM Instruct)",
   "reasons": [],
   "summary": "HELM Instruct scenario over OASST1 initial prompts, scored with a 1-5 Helpfulness critique rather than gold replies."
  },
  {
   "aliases": [
    "openai/mrcr",
    "OpenAIMRCRScenario",
    "Multi-round co-reference resolution"
   ],
   "canonical_id": "openai_mrcr",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "openai_mrcr",
   "models_covered": 0,
   "name": "OpenAI MRCR",
   "reasons": [],
   "summary": "OpenAI's MRCR: reproduce the i-th matching assistant writing in a long synthetic chat, after a hash prefix."
  },
  {
   "aliases": [],
   "canonical_id": "openbookqa",
   "category": "reasoning",
   "disposition": "unverified",
   "id": "openbookqa",
   "models_covered": 0,
   "name": "OpenBookQA",
   "reasons": [
    "no qualifying current frontier/open coverage from different organizations",
    "review is not approved",
    "task, metric, and protocol are incomplete",
    "usefulness is not established",
    "usefulness is unknown"
   ],
   "summary": "5,957 four-way elementary-science questions built around a small fact book; human accuracy is near 92%, and no current model card or live leaderboard still reports it."
  },
  {
   "aliases": [
    "OpenFinData_gen",
    "openfindata_release"
   ],
   "canonical_id": "openfindata",
   "category": "domain",
   "disposition": "unassessed",
   "id": "openfindata",
   "models_covered": 0,
   "name": "OpenFinData",
   "reasons": [],
   "summary": "1,500 Chinese financial items from East Money scenes; OpenCompass scores nine of 19 files (650 items) with letter accuracy or keyword hit."
  },
  {
   "aliases": [],
   "canonical_id": "openml_benchmark",
   "category": "knowledge",
   "disposition": "unverified",
   "id": "openml_benchmark",
   "models_covered": 0,
   "name": "OpenML Benchmark",
   "reasons": [
    "no qualifying current frontier/open coverage from different organizations",
    "review is not approved",
    "task, metric, and protocol are incomplete",
    "usefulness is not established",
    "usefulness is unknown"
   ],
   "summary": "OpenML Benchmark is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "openml_benchmarks",
   "category": "knowledge",
   "disposition": "unverified",
   "id": "openml_benchmarks",
   "models_covered": 0,
   "name": "OpenML Benchmarks",
   "reasons": [
    "no qualifying current frontier/open coverage from different organizations",
    "review is not approved",
    "task, metric, and protocol are incomplete",
    "usefulness is not established",
    "usefulness is unknown"
   ],
   "summary": "OpenML Benchmarks is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "openml_explain",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "openml_explain",
   "models_covered": 0,
   "name": "OpenML Explain",
   "reasons": [],
   "summary": "OpenML Explain is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [
    "OpenSWI-shallow",
    "OpenSWI-deep",
    "OpenSWI-real",
    "openswi_gen"
   ],
   "canonical_id": "openswi",
   "category": "domain",
   "disposition": "unassessed",
   "id": "openswi",
   "models_covered": 0,
   "name": "OpenSWI",
   "reasons": [],
   "summary": "Geophysics inversion set of ~23 million Vs\u2013dispersion pairs; OpenCompass scores two LLM configs named OpenSWI-shallow-1k and OpenSWI-deep-1k with RMSE."
  },
  {
   "aliases": [
    "BIG-bench operators",
    "BIG-bench Lite operators"
   ],
   "canonical_id": "operators",
   "category": "math",
   "disposition": "unassessed",
   "id": "operators",
   "models_covered": 0,
   "name": "Operators",
   "reasons": [],
   "summary": "A 211-item BIG-bench Lite task: read an English definition of a novel operator and return the numeric result, zero-shot."
  },
  {
   "aliases": [
    "OpinionsQA",
    "opinions_qa",
    "Opinion QA"
   ],
   "canonical_id": "opinions_qa",
   "category": "safety",
   "disposition": "unassessed",
   "id": "opinions_qa",
   "models_covered": 0,
   "name": "OpinionQA",
   "reasons": [],
   "summary": "1,498 Pew American Trends Panel multiple-choice opinion questions used to compare a model's answer distribution with 60 US demographic groups, not to score accuracy."
  },
  {
   "aliases": [],
   "canonical_id": "opt_bench",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "opt_bench",
   "models_covered": 0,
   "name": "OPT-BENCH",
   "reasons": [],
   "summary": "OPT-BENCH is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "opt_engine",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "opt_engine",
   "models_covered": 0,
   "name": "OPT-Engine",
   "reasons": [],
   "summary": "OPT-Engine is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [
    "OS-World"
   ],
   "canonical_id": "osworld",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "osworld",
   "models_covered": 3,
   "name": "OSWorld",
   "reasons": [],
   "summary": "Tests whether a multimodal agent can complete open-ended tasks in a real, live desktop operating system."
  },
  {
   "aliases": [
    "Perplexity Analysis for Language Model Assessment",
    "PALOMA"
   ],
   "canonical_id": "paloma",
   "category": "generation",
   "disposition": "unassessed",
   "id": "paloma",
   "models_covered": 0,
   "name": "Paloma",
   "reasons": [],
   "summary": "Allen AI fit benchmark over 546 English and code domains from 16 sources, scored as perplexity and bits per byte rather than task accuracy."
  },
  {
   "aliases": [
    "PaperBench",
    "PaperBench Code-Dev"
   ],
   "canonical_id": "paperbench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "paperbench",
   "models_covered": 0,
   "name": "PaperBench",
   "reasons": [],
   "summary": "Agents must replicate 20 ICML 2024 Spotlight and Oral papers from scratch against author-written rubrics totaling 8,316 leaf criteria."
  },
  {
   "aliases": [
    "BIG-bench paragraph_segmentation"
   ],
   "canonical_id": "paragraph_segmentation",
   "category": "generation",
   "disposition": "unassessed",
   "id": "paragraph_segmentation",
   "models_covered": 0,
   "name": "Paragraph Segmentation",
   "reasons": [],
   "summary": "A 9,000-document BIG-bench task: label which sentences end paragraphs in nine European languages as a 0/1 sequence."
  },
  {
   "aliases": [
    "parsinlu_qa",
    "ParsiNLU QA",
    "ParsiNLU multiple-choice QA"
   ],
   "canonical_id": "parsinlu_qa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "parsinlu_qa",
   "models_covered": 0,
   "name": "PARSINLU QA",
   "reasons": [],
   "summary": "1,050 open-domain Persian multiple-choice questions from Iranian exams, a BIG-bench task drawn from the ParsiNLU suite."
  },
  {
   "aliases": [
    "BIG-bench parsinlu_reading_comprehension",
    "ParsiNLU RC"
   ],
   "canonical_id": "parsinlu_reading_comprehension",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "parsinlu_reading_comprehension",
   "models_covered": 0,
   "name": "ParsiNLU Reading Comprehension",
   "reasons": [],
   "summary": "A 518-item Persian span-extraction BIG-bench Lite task taken from the ParsiNLU reading-comprehension eval split."
  },
  {
   "aliases": [
    "PAWS-Wiki",
    "Paraphrase Adversaries from Word Scrambling"
   ],
   "canonical_id": "paws",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "paws",
   "models_covered": 0,
   "name": "PAWS",
   "reasons": [],
   "summary": "Inspect Evals yes/no paraphrase detection on the 8,000-item PAWS-Wiki labeled-final test set of high-overlap sentence pairs."
  },
  {
   "aliases": [
    "paws-x",
    "pawsx",
    "PAWS-X: A Cross-lingual Adversarial Dataset for Paraphrase Identification"
   ],
   "canonical_id": "paws_x",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "paws_x",
   "models_covered": 0,
   "name": "PAWS-X",
   "reasons": [],
   "summary": "lm-eval multilingual paraphrase identification on PAWS-X: seven languages of high-overlap sentence pairs, scored by accuracy."
  },
  {
   "aliases": [
    "Tables of Penguins",
    "BIG-bench penguins_in_a_table",
    "BBH penguins_in_a_table"
   ],
   "canonical_id": "penguins_in_a_table",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "penguins_in_a_table",
   "models_covered": 0,
   "name": "Penguins in a Table",
   "reasons": [],
   "summary": "A 149-item BIG-bench table-QA task about named penguins; BIG-bench Hard keeps 146 of the same items."
  },
  {
   "aliases": [
    "QML Benchmarks",
    "qml-benchmarks"
   ],
   "canonical_id": "pennylane_qml",
   "category": "domain",
   "disposition": "unassessed",
   "id": "pennylane_qml",
   "models_covered": 0,
   "name": "PennyLane QML Benchmarks",
   "reasons": [],
   "summary": "Twelve PennyLane datasets from two Xanadu studies that score quantum classifiers and generators against classical models on synthetic and real bit-string tasks."
  },
  {
   "aliases": [
    "BIG-bench periodic_elements"
   ],
   "canonical_id": "periodic_elements",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "periodic_elements",
   "models_covered": 0,
   "name": "Periodic Elements (BIG-bench)",
   "reasons": [],
   "summary": "A 654-item BIG-bench chemistry task: name the element from its atomic number, or from a one-step neighbour on the table."
  },
  {
   "aliases": [
    "BIG-bench persian_idioms"
   ],
   "canonical_id": "persian_idioms",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "persian_idioms",
   "models_covered": 0,
   "name": "Persian Idioms (BIG-bench)",
   "reasons": [],
   "summary": "A 66-item BIG-bench task that picks the conventional meaning of a Persian idiom, with Persian and English four-way choices."
  },
  {
   "aliases": [
    "PersistBench",
    "persistbench_cross_domain",
    "persistbench_sycophancy",
    "persistbench_beneficial_memory"
   ],
   "canonical_id": "persistbench",
   "category": "safety",
   "disposition": "unassessed",
   "id": "persistbench",
   "models_covered": 0,
   "name": "PersistBench",
   "reasons": [],
   "summary": "500 memory-and-query samples that score whether a model leaks, sycophantically agrees with, or correctly uses injected long-term user memories."
  },
  {
   "aliases": [
    "inspect_evals personality",
    "personality_BFI",
    "personality_TRAIT"
   ],
   "canonical_id": "personality",
   "category": "domain",
   "disposition": "unassessed",
   "id": "personality",
   "models_covered": 0,
   "name": "Personality (Inspect Evals)",
   "reasons": [],
   "summary": "Inspect Evals suite that scores an LLM's Big Five and Dark Triad profile from BFI and TRAIT questionnaires, not factual accuracy."
  },
  {
   "aliases": [],
   "canonical_id": "perspectivegap",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "perspectivegap",
   "models_covered": 0,
   "name": "PerspectiveGap",
   "reasons": [],
   "summary": "110 scenarios testing whether a model can write orchestration prompts that give each sub-agent in a multi-agent system exactly what it needs to know, without leaking irrelevant context."
  },
  {
   "aliases": [
    "BIG-bench phrase_relatedness"
   ],
   "canonical_id": "phrase_relatedness",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "phrase_relatedness",
   "models_covered": 0,
   "name": "Phrase Relatedness (BIG-bench)",
   "reasons": [],
   "summary": "A 100-item BIG-bench task: given a short phrase, pick the most semantically related n-gram among four choices."
  },
  {
   "aliases": [
    "PHYBench",
    "PhyBench"
   ],
   "canonical_id": "phybench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "phybench",
   "models_covered": 0,
   "name": "PHYBench",
   "reasons": [],
   "summary": "500 original text-only physics problems scored by expression-tree edit distance; Gemini 2.5 Pro reached 36.9% accuracy versus a 61.9% human baseline."
  },
  {
   "aliases": [
    "BIG-bench physical_intuition"
   ],
   "canonical_id": "physical_intuition",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "physical_intuition",
   "models_covered": 0,
   "name": "Physical Intuition (BIG-bench)",
   "reasons": [],
   "summary": "An 81-item BIG-bench task that picks the dominant physical mechanism or behaviour of a described system."
  },
  {
   "aliases": [
    "PHYSICS Benchmark"
   ],
   "canonical_id": "physics",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "physics",
   "models_covered": 0,
   "name": "PHYSICS (Benchmarking Foundation Models on University-Level Physics Problem Solving)",
   "reasons": [],
   "summary": "1,297 PhD-qualifying-exam physics problems across six subfields, needing multi-step derivation; the best model in the 2025 paper solved only 59.9% of the test set."
  },
  {
   "aliases": [
    "physics_gre_multiple_choice",
    "GR8677",
    "Inflection Physics GRE"
   ],
   "canonical_id": "physics_gre",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "physics_gre",
   "models_covered": 0,
   "name": "Physics GRE (Inflection-Benchmarks)",
   "reasons": [],
   "summary": "Processed Physics GRE exams scored as five-way multiple choice; lm-eval reports accuracy on image-free items, not Inflection's GRE percentile."
  },
  {
   "aliases": [
    "Physics Questions"
   ],
   "canonical_id": "physics_questions",
   "category": "math",
   "disposition": "unassessed",
   "id": "physics_questions",
   "models_covered": 0,
   "name": "physics_questions (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench task of 53 edited high-school physics word problems scored by exact string match on a number-plus-unit answer."
  },
  {
   "aliases": [
    "PI-LLM Bench",
    "PI_LLM",
    "pi-llm"
   ],
   "canonical_id": "pi_llm",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "pi_llm",
   "models_covered": 0,
   "name": "PI-LLM",
   "reasons": [],
   "summary": "A key-value overwrite test of proactive interference: the model must report each key's last value after many similar updates, isolating working-memory limits inside the context window."
  },
  {
   "aliases": [
    "Pile BPB",
    "pile bits-per-byte"
   ],
   "canonical_id": "pile",
   "category": "generation",
   "disposition": "unassessed",
   "id": "pile",
   "models_covered": 0,
   "name": "The Pile (lm-eval BPB group)",
   "reasons": [],
   "summary": "lm-eval group that scores bits-per-byte and perplexity on 22 Pile component streams as a language-modelling eval, not a QA task."
  },
  {
   "aliases": [
    "pile-10k",
    "NeelNanda/pile-10k"
   ],
   "canonical_id": "pile_10k",
   "category": "generation",
   "disposition": "unassessed",
   "id": "pile_10k",
   "models_covered": 0,
   "name": "Pile-10k",
   "reasons": [],
   "summary": "Rolling loglikelihood on the first 10,000 Pile documents; a debug sample, not the official Pile test split."
  },
  {
   "aliases": [
    "Physical Interaction QA",
    "Physical IQA"
   ],
   "canonical_id": "piqa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "piqa",
   "models_covered": 0,
   "name": "PIQA",
   "reasons": [],
   "summary": "Binary-choice physical commonsense reasoning built from instructables.com how-to text; a 2019 benchmark now close to its human baseline for most current models."
  },
  {
   "aliases": [
    "PisaBench",
    "pisa-bench",
    "PISA-Bench (lm-eval pisa)"
   ],
   "canonical_id": "pisa",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "pisa",
   "models_covered": 0,
   "name": "PISA-Bench",
   "reasons": [],
   "summary": "A six-language VLM benchmark of ~122 OECD PISA exam items with images, scored as accuracy in lm-eval groups pisa and pisa_llm_judged."
  },
  {
   "aliases": [
    "PJExamDataset"
   ],
   "canonical_id": "pjexam",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "pjexam",
   "models_covered": 0,
   "name": "PJExam",
   "reasons": [],
   "summary": "An OpenCompass Chinese exam suite of 2022-2023 Gaokao and 2022 Zhongkao multiple-choice items, scored from a chain-of-thought answer letter; the public dataset class is missing."
  },
  {
   "aliases": [
    "Shakespeare Dialogue",
    "BIG-bench play_dialog_same_or_different"
   ],
   "canonical_id": "play_dialog_same_or_different",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "play_dialog_same_or_different",
   "models_covered": 0,
   "name": "Play Dialogue Same or Different (BIG-bench)",
   "reasons": [],
   "summary": "A 3,264-item BIG-bench Lite task: decide whether two nearby Shakespeare-play lines were spoken by the same character."
  },
  {
   "aliases": [
    "PMC-Patients ReCDS",
    "ReCDS-PAR",
    "ReCDS-PPR"
   ],
   "canonical_id": "pmc_patients",
   "category": "embedding",
   "disposition": "unassessed",
   "id": "pmc_patients",
   "models_covered": 0,
   "name": "PMC-Patients",
   "reasons": [],
   "summary": "167k PMC case-report summaries with citation-graph labels for retrieving relevant PubMed articles (PAR) and similar patients (PPR)."
  },
  {
   "aliases": [
    "PMMEval",
    "P MMEval"
   ],
   "canonical_id": "pmmeval",
   "category": "composite",
   "disposition": "unassessed",
   "id": "pmmeval",
   "models_covered": 0,
   "name": "P-MMEval",
   "reasons": [],
   "summary": "A Qwen/Tongyi parallel multilingual suite that extends eight existing tasks across the same ten languages so cross-lingual gaps are not confounded with different item sets."
  },
  {
   "aliases": [
    "PolEmo2",
    "PolEmo 2.0",
    "klej-polemo2",
    "polemo2_in",
    "polemo2_out"
   ],
   "canonical_id": "polemo2",
   "category": "domain",
   "disposition": "unassessed",
   "id": "polemo2",
   "models_covered": 0,
   "name": "PolEmo 2.0",
   "reasons": [],
   "summary": "Polish four-class review sentiment from PolEmo 2.0, scored in-domain and out-of-domain as two lm-evaluation-harness tasks."
  },
  {
   "aliases": [
    "Sequence Labeling",
    "BIG-bench polish_sequence_labeling"
   ],
   "canonical_id": "polish_sequence_labeling",
   "category": "domain",
   "disposition": "unassessed",
   "id": "polish_sequence_labeling",
   "models_covered": 0,
   "name": "Polish Sequence Labeling (BIG-bench)",
   "reasons": [],
   "summary": "A BIG-bench Polish sequence-labeling task: emit NER, time, and event tags for each token of a KPWr sentence."
  },
  {
   "aliases": [
    "portuguese_bench",
    "Portuguese Bench"
   ],
   "canonical_id": "portuguese_bench",
   "category": "composite",
   "disposition": "unassessed",
   "id": "portuguese_bench",
   "models_covered": 0,
   "name": "PortugueseBench",
   "reasons": [],
   "summary": "IberoBench's European Portuguese suite: ASSIN entailment and paraphrase, Belebele reading, and FLORES translation, plus two extra ASSIN2 tasks in the current harness group."
  },
  {
   "aliases": [
    "Pre-Flight",
    "inspect_evals/pre_flight",
    "pre-flight-06"
   ],
   "canonical_id": "pre_flight",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "pre_flight",
   "models_covered": 0,
   "name": "Pre-Flight",
   "reasons": [],
   "summary": "300 multiple-choice questions on ICAO, FAA, and airport ground-ops knowledge, scored by Inspect Evals accuracy."
  },
  {
   "aliases": [
    "Presuppositions as Natural Language Inference",
    "BIG-bench presuppositions_as_nli"
   ],
   "canonical_id": "presuppositions_as_nli",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "presuppositions_as_nli",
   "models_covered": 0,
   "name": "Presuppositions as NLI (BIG-bench)",
   "reasons": [],
   "summary": "A 735-item BIG-bench three-way NLI task on whether English sentences presuppose a given hypothesis, including negated and adversarial pairs."
  },
  {
   "aliases": [
    "Phone Realization in Speech Models"
   ],
   "canonical_id": "prism",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "prism",
   "models_covered": 0,
   "name": "PRiSM",
   "reasons": [],
   "summary": "Open suite for phone recognition: phonetic-feature error on IPA transcripts plus clinical, L2, and multilingual probes of speech models."
  },
  {
   "aliases": [],
   "canonical_id": "prism_bench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "prism_bench",
   "models_covered": 0,
   "name": "PRISM-Bench",
   "reasons": [],
   "summary": "900 human-checked clips that score text-to-audio-video models on audio type and on-screen vs off-screen sound with an MLLM judge."
  },
  {
   "aliases": [],
   "canonical_id": "processbench",
   "category": "math",
   "disposition": "unassessed",
   "id": "processbench",
   "models_covered": 0,
   "name": "ProcessBench",
   "reasons": [],
   "summary": "3,400 human-annotated math solutions, mostly contest-level, where the model must name the earliest wrong step or report that the trace is clean."
  },
  {
   "aliases": [
    "Python Program Synthesis",
    "BIG-bench program_synthesis"
   ],
   "canonical_id": "program_synthesis",
   "category": "coding",
   "disposition": "unassessed",
   "id": "program_synthesis",
   "models_covered": 0,
   "name": "Python Program Synthesis (BIG-bench)",
   "reasons": [],
   "summary": "A BIG-bench programmatic task: write the simplest Python f(x) that fits noisy input/output pairs, scored for compile, correctness, and length."
  },
  {
   "aliases": [
    "PromptRobust",
    "Prompt Bench"
   ],
   "canonical_id": "promptbench",
   "category": "safety",
   "disposition": "unassessed",
   "id": "promptbench",
   "models_covered": 0,
   "name": "PromptBench",
   "reasons": [],
   "summary": "Microsoft's adversarial-prompt robustness eval: character- to semantic-level attacks on the instruction, scored as the drop on GLUE, MMLU, SQuAD, translation and math."
  },
  {
   "aliases": [
    "FormalProofBench",
    "Proof Bench",
    "ProofBench v1.1"
   ],
   "canonical_id": "proofbench",
   "category": "math",
   "disposition": "unassessed",
   "id": "proofbench",
   "models_covered": 0,
   "name": "ProofBench",
   "reasons": [],
   "summary": "Private 100-problem Lean 4 suite: given a natural-language statement and a formal theorem, the model must write a proof the checker accepts."
  },
  {
   "aliases": [
    "PROST",
    "Physical Reasoning about Objects Through Space and Time",
    "corypaik/prost"
   ],
   "canonical_id": "prost",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "prost",
   "models_covered": 0,
   "name": "PROST",
   "reasons": [],
   "summary": "18,736 four-choice English questions from 14 templates that probe physical reasoning about objects in space and time, meant to be used zero-shot."
  },
  {
   "aliases": [
    "Protein Interaction Sites",
    "BIG-bench protein_interacting_sites"
   ],
   "canonical_id": "protein_interacting_sites",
   "category": "domain",
   "disposition": "unassessed",
   "id": "protein_interacting_sites",
   "models_covered": 0,
   "name": "Protein Interacting Sites (BIG-bench)",
   "reasons": [],
   "summary": "A BIG-bench programmatic probe: decide whether a named amino acid in a spelled-out protein sequence is an interacting residue."
  },
  {
   "aliases": [],
   "canonical_id": "proteinlmbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "proteinlmbench",
   "models_covered": 0,
   "name": "ProteinLMBench",
   "reasons": [],
   "summary": "944 manually checked multiple-choice questions on proteins, used to score LLMs on sequence and function understanding rather than to train them."
  },
  {
   "aliases": [
    "PQA-L"
   ],
   "canonical_id": "pubmedqa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "pubmedqa",
   "models_covered": 4,
   "name": "PubMedQA",
   "reasons": [],
   "summary": "Yes/no/maybe research questions answered from their source PubMed abstract, testing biomedical reading comprehension."
  },
  {
   "aliases": [
    "Putnam AXIOM",
    "Putnam-AXIOM Original",
    "putnam_axiom_original"
   ],
   "canonical_id": "putnam_axiom",
   "category": "math",
   "disposition": "unassessed",
   "id": "putnam_axiom",
   "models_covered": 0,
   "name": "Putnam-AXIOM",
   "reasons": [],
   "summary": "522 Putnam contest problems scored by boxed exact match, plus 100 functional variants meant to catch memorisation of public contest write-ups."
  },
  {
   "aliases": [
    "py150",
    "CodeXGLUE PY150 line completion"
   ],
   "canonical_id": "py150",
   "category": "coding",
   "disposition": "unassessed",
   "id": "py150",
   "models_covered": 0,
   "name": "PY150 (OpenCompass line completion)",
   "reasons": [],
   "summary": "OpenCompass wrap of CodeXGLUE's PY150 line-level completion: given a Python prefix, generate the next line and score it with BLEU."
  },
  {
   "aliases": [
    "Python Programming",
    "python programming challenge"
   ],
   "canonical_id": "python_programming_challenge",
   "category": "coding",
   "disposition": "unassessed",
   "id": "python_programming_challenge",
   "models_covered": 0,
   "name": "Python Programming Challenge",
   "reasons": [],
   "summary": "A 32-challenge BIG-bench task that asks a model to complete a Python function, then compiles and unit-tests the body in a RestrictedPython sandbox."
  },
  {
   "aliases": [
    "Question Answering for Machine Reading Evaluation",
    "qa4mre_2011",
    "qa4mre_2012",
    "qa4mre_2013"
   ],
   "canonical_id": "qa4mre",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "qa4mre",
   "models_covered": 0,
   "name": "QA4MRE",
   "reasons": [],
   "summary": "CLEF 2011\u20132013 reading-comprehension lab: five-way questions about one document, shipped in lm-eval as English main-track sets for each year."
  },
  {
   "aliases": [
    "qa_wikidata",
    "QA Wikidata"
   ],
   "canonical_id": "qa_wikidata",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "qa_wikidata",
   "models_covered": 0,
   "name": "QA WikiData",
   "reasons": [],
   "summary": "BIG-bench JSON task of 20,442 English cloze prompts built from Wikidata triples, scored with ROUGE-Lsum."
  },
  {
   "aliases": [],
   "canonical_id": "qabench",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "qabench",
   "models_covered": 0,
   "name": "qabench",
   "reasons": [],
   "summary": "An OpenCompass generation config over a local qabench-test.qa.csv of prompt/reference pairs; the public tree does not identify a paper, licence, or item count."
  },
  {
   "aliases": [
    "Question Answering over Scientific Papers"
   ],
   "canonical_id": "qasper",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "qasper",
   "models_covered": 0,
   "name": "QASPER",
   "reasons": [],
   "summary": "QASPER pairs 5,049 questions, written from only a title and abstract, with 1,585 full NLP papers whose text must supply the answer, making it a long-context benchmark by design."
  },
  {
   "aliases": [
    "qaspercut",
    "QASPERCUT"
   ],
   "canonical_id": "qaspercut",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "qaspercut",
   "models_covered": 0,
   "name": "QASPER-cut",
   "reasons": [],
   "summary": "OpenCompass QASPER variant that keeps extractive-span questions and feeds the paper text from the first gold evidence offset onward."
  },
  {
   "aliases": [
    "QuAC",
    "Question Answering in Context"
   ],
   "canonical_id": "quac",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "quac",
   "models_covered": 0,
   "name": "QuAC (Question Answering in Context)",
   "reasons": [],
   "summary": "Student\u2013teacher dialogs over a hidden Wikipedia section; HELM scores free-text F1 at a random turn, while the official board uses span F1 on a hidden test set."
  },
  {
   "aliases": [
    "QuALITY: Question Answering with Long Input Texts, Yes!",
    "QUALITY"
   ],
   "canonical_id": "quality",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "quality",
   "models_covered": 0,
   "name": "QuALITY",
   "reasons": [],
   "summary": "Multiple-choice QA over English passages averaging about 5,000 tokens; writers read the full article, and a hard subset beats speed-limited annotators."
  },
  {
   "aliases": [
    "question_answer_creation",
    "QA creation from COPA"
   ],
   "canonical_id": "question_answer_creation",
   "category": "generation",
   "disposition": "unassessed",
   "id": "question_answer_creation",
   "models_covered": 0,
   "name": "Question-Answer Creation",
   "reasons": [],
   "summary": "Programmatic BIG-bench task that asks a model to write new COPA-style multiple-choice items and then answer them; the score is validity times self-consistency."
  },
  {
   "aliases": [
    "question selection"
   ],
   "canonical_id": "question_selection",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "question_selection",
   "models_covered": 0,
   "name": "Question Selection (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench task: given an English short answer and its paragraph, choose which paraphrased question that answer actually satisfies."
  },
  {
   "aliases": [
    "RBench",
    "Reasoning Bench"
   ],
   "canonical_id": "r_bench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "r_bench",
   "models_covered": 0,
   "name": "R-Bench (Reasoning Bench)",
   "reasons": [],
   "summary": "A graduate-level, English/Chinese multi-disciplinary reasoning benchmark - 1,094 text questions across 108 subjects and 665 multimodal questions across 83 subjects, calibrated for Olympiad-level difficulty."
  },
  {
   "aliases": [
    "ReAding Comprehension dataset from Examinations"
   ],
   "canonical_id": "race",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "race",
   "models_covered": 0,
   "name": "RACE",
   "reasons": [],
   "summary": "RACE is multiple-choice reading comprehension from English exams written for Chinese middle- and high-school students, split into easier RACE-M and harder RACE-H passages, four options per question."
  },
  {
   "aliases": [
    "RaceBias",
    "race-based med",
    "RaceBasedMedScenario"
   ],
   "canonical_id": "race_based_med",
   "category": "safety",
   "disposition": "unassessed",
   "id": "race_based_med",
   "models_covered": 0,
   "name": "RaceBias (HELM race_based_med)",
   "reasons": [],
   "summary": "HELM MedHELM task: read a medical question and a stored model answer, then say whether that answer contains race-based, harmful, or inaccurate content."
  },
  {
   "aliases": [
    "RACE-H",
    "race-high"
   ],
   "canonical_id": "race_h",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "race_h",
   "models_covered": 0,
   "name": "RACE-H (inspect_evals)",
   "reasons": [],
   "summary": "inspect_evals task for RACE-H: pick one of four answers to a high-school English-exam question about a passage."
  },
  {
   "aliases": [
    "Real-world Annotated Few-shot Tasks",
    "ought/raft"
   ],
   "canonical_id": "raft",
   "category": "domain",
   "disposition": "unassessed",
   "id": "raft",
   "models_covered": 0,
   "name": "RAFT (Real-world Annotated Few-shot Tasks)",
   "reasons": [],
   "summary": "Eleven real-world English classification tasks with 50 labeled examples each; official score is macro-F1 on a hidden test set."
  },
  {
   "aliases": [
    "RE-Bench: A Multilingual Benchmark for Issue Resolving"
   ],
   "canonical_id": "re_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "re_bench",
   "models_covered": 0,
   "name": "RE-Bench",
   "reasons": [],
   "summary": "RE-Bench evaluates agents that modify repositories to resolve issues across Java, TypeScript, JavaScript, Go, Rust, C and C++."
  },
  {
   "aliases": [
    "RoFT",
    "Real or Fake Text"
   ],
   "canonical_id": "real_or_fake_text",
   "category": "generation",
   "disposition": "unassessed",
   "id": "real_or_fake_text",
   "models_covered": 0,
   "name": "Real or Fake Text (RoFT, BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench RoFT task: a 10-sentence English passage starts human-written; pick the first machine-generated sentence."
  },
  {
   "aliases": [
    "Real Toxicity Prompts",
    "realtoxicprompts",
    "real-toxicity-prompts",
    "RTP"
   ],
   "canonical_id": "real_toxicity_prompts",
   "category": "safety",
   "disposition": "unassessed",
   "id": "real_toxicity_prompts",
   "models_covered": 0,
   "name": "RealToxicityPrompts",
   "reasons": [],
   "summary": "English prompt-continuation toxicity benchmark: models continue web-text prefixes; PerspectiveAPI scores how toxic the completions are."
  },
  {
   "aliases": [],
   "canonical_id": "realworldqa",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "realworldqa",
   "models_covered": 8,
   "name": "RealWorldQA",
   "reasons": [],
   "summary": "A multiple-choice test of everyday spatial understanding built from over 700 real-world photos, each paired with one verifiable question."
  },
  {
   "aliases": [
    "BBH reasoning_about_colored_objects"
   ],
   "canonical_id": "reasoning_about_colored_objects",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "reasoning_about_colored_objects",
   "models_covered": 0,
   "name": "Reasoning about Colored Objects (BIG-bench)",
   "reasons": [],
   "summary": "2,000 generated BIG-bench questions about colors, left/right order, and counts of household objects; BIG-bench Hard keeps 250."
  },
  {
   "aliases": [
    "BIG-bench Lite repeat_copy_logic"
   ],
   "canonical_id": "repeat_copy_logic",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "repeat_copy_logic",
   "models_covered": 0,
   "name": "Repeat Copy Logic (BIG-bench)",
   "reasons": [],
   "summary": "32 English instructions that copy, repeat, and insert words; a BIG-bench Lite exact-string-match task."
  },
  {
   "aliases": [
    "Keyword Sentence Transformation"
   ],
   "canonical_id": "rephrase",
   "category": "generation",
   "disposition": "unassessed",
   "id": "rephrase",
   "models_covered": 0,
   "name": "Rephrase (BIG-bench)",
   "reasons": [],
   "summary": "78 English keyword-transformation items: rewrite a sentence so it keeps the meaning and contains a fixed keyword."
  },
  {
   "aliases": [],
   "canonical_id": "rhyming",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "rhyming",
   "models_covered": 0,
   "name": "Rhyming (BIG-bench)",
   "reasons": [],
   "summary": "Two BIG-bench English subtasks: 680 five-way rhyme picks from CMUdict, and 273 verse rhyme-scheme strings."
  },
  {
   "aliases": [
    "BIG-bench riddle_sense"
   ],
   "canonical_id": "riddle_sense",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "riddle_sense",
   "models_covered": 0,
   "name": "RiddleSense (BIG-bench)",
   "reasons": [],
   "summary": "49 five-way riddle questions shipped in BIG-bench; not the full 5,715-item RiddleSense dataset."
  },
  {
   "aliases": [
    "Ro-Bench",
    "Ro-Bench: Robust Video MLLMs Benchmark"
   ],
   "canonical_id": "ro_bench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "ro_bench",
   "models_covered": 0,
   "name": "RO-Bench",
   "reasons": [],
   "summary": "Multiple-choice video questions scored on original clips and on text-edited counterfactual versions, so the gap measures robustness rather than raw accuracy."
  },
  {
   "aliases": [],
   "canonical_id": "ro_n3ws",
   "category": "domain",
   "disposition": "unassessed",
   "id": "ro_n3ws",
   "models_covered": 0,
   "name": "RO-N3WS",
   "reasons": [],
   "summary": "A 126-hour Romanian ASR set that trains on broadcast news and tests on news plus audiobook, film, story, and podcast speech, scored by word error rate."
  },
  {
   "aliases": [
    "RoleLLM",
    "RoleBench (RoleLLM)"
   ],
   "canonical_id": "rolebench",
   "category": "generation",
   "disposition": "unassessed",
   "id": "rolebench",
   "models_covered": 0,
   "name": "RoleBench",
   "reasons": [],
   "summary": "A 168,093-sample role-play benchmark over 100 characters; OpenCompass scores English and Chinese splits with ROUGE against RoleGPT-style references."
  },
  {
   "aliases": [
    "Root Finding, Optimization and Games"
   ],
   "canonical_id": "roots_optimization_and_games",
   "category": "math",
   "disposition": "unassessed",
   "id": "roots_optimization_and_games",
   "models_covered": 0,
   "name": "Roots, Optimization and Games (BIG-bench)",
   "reasons": [],
   "summary": "Programmatic BIG-bench suite of integer-root polynomials, 1-D convex minima, and two-action payoff questions, scored as mean accuracy full."
  },
  {
   "aliases": [
    "Ruin a Name with One Edit",
    "BBH ruin_names"
   ],
   "canonical_id": "ruin_names",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "ruin_names",
   "models_covered": 0,
   "name": "Ruin Names (BIG-bench)",
   "reasons": [],
   "summary": "448 four-way questions: pick the funny one-character edit of a movie or band name; BIG-bench Hard keeps 250."
  },
  {
   "aliases": [
    "RULER: What's the Real Context Size of Your Long-Context Language Models?"
   ],
   "canonical_id": "ruler",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "ruler",
   "models_covered": 0,
   "name": "RULER",
   "reasons": [],
   "summary": "A synthetic long-context suite of 13 tasks across four categories, built to show a model's effective context length is usually shorter than its claimed maximum."
  },
  {
   "aliases": [
    "S\u00b2-Bench",
    "S2-Bench",
    "TOMG-Bench",
    "Speak-to-Structure"
   ],
   "canonical_id": "s2_tomg_bench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "s2_tomg_bench",
   "models_covered": 0,
   "name": "S2-TOMG-Bench",
   "reasons": [],
   "summary": "Nine chemistry generation subtasks (45,000 instructions) that ask for SMILES molecules satisfying atom, edit, or property constraints, scored with RDKit success and weighted success."
  },
  {
   "aliases": [
    "S3Eval: A Synthetic, Scalable, Systematic Evaluation Suite for Large Language Models"
   ],
   "canonical_id": "s3eval",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "s3eval",
   "models_covered": 0,
   "name": "S3Eval",
   "reasons": [],
   "summary": "A synthetic SQL-execution benchmark that generates unlimited, contamination-resistant tables and queries to probe reasoning and long-context comprehension from 200 tokens to 200K."
  },
  {
   "aliases": [
    "Situational Awareness Dataset",
    "SAD-mini",
    "Me, Myself, and AI"
   ],
   "canonical_id": "sad",
   "category": "safety",
   "disposition": "unassessed",
   "id": "sad",
   "models_covered": 0,
   "name": "SAD (Situational Awareness Dataset)",
   "reasons": [],
   "summary": "Inspect Evals SAD-mini: 2,904 multiple-choice items in five tasks on whether a model knows it is an LLM and can place itself in training, evaluation, or deployment."
  },
  {
   "aliases": [
    "safety_gen",
    "safety_datasets"
   ],
   "canonical_id": "safety",
   "category": "safety",
   "disposition": "unassessed",
   "id": "safety",
   "models_covered": 0,
   "name": "OpenCompass safety (Perspective toxicity)",
   "reasons": [],
   "summary": "OpenCompass runs free-text completions on prompts from a local safety.txt file and scores toxicity with Google Perspective API."
  },
  {
   "aliases": [
    "Sage",
    "Scientific AGentic retrieval Evaluation"
   ],
   "canonical_id": "sage",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "sage",
   "models_covered": 0,
   "name": "SAGE",
   "reasons": [],
   "summary": "1,200 scientific-literature queries over a closed paper corpus that tests whether deep-research agents retrieve the right papers, not just browse the open web."
  },
  {
   "aliases": [
    "Salient Translation Error Detection"
   ],
   "canonical_id": "salient_translation_error_detection",
   "category": "translation",
   "disposition": "unassessed",
   "id": "salient_translation_error_detection",
   "models_covered": 0,
   "name": "Salient Translation Error Detection (BIG-bench)",
   "reasons": [],
   "summary": "A 998-item BIG-bench multiple-choice task: name the error type in an English translation of a German sentence."
  },
  {
   "aliases": [
    "scBench: Evaluating AI Agents on Single-Cell RNA-seq Analysis",
    "scbench"
   ],
   "canonical_id": "scbench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "scbench",
   "models_covered": 0,
   "name": "scBench",
   "reasons": [],
   "summary": "An agent benchmark of verifiable scRNA-seq analysis problems; inspect_evals ships a 30-task public canonical slice, while the paper describes 394 held-out tasks."
  },
  {
   "aliases": [
    "SciBench: Evaluating College-Level Scientific Problem-Solving Abilities of Large Language Models"
   ],
   "canonical_id": "scibench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "scibench",
   "models_covered": 0,
   "name": "SciBench",
   "reasons": [],
   "summary": "College-level chemistry, physics and math word problems from textbooks; OpenCompass scores ten text subsets by numeric exact match after a boxed-answer parse."
  },
  {
   "aliases": [
    "SciCode: A Research Coding Benchmark Curated by Scientists"
   ],
   "canonical_id": "scicode",
   "category": "coding",
   "disposition": "active",
   "id": "scicode",
   "models_covered": 0,
   "name": "SciCode",
   "reasons": [
    "qualifying current coverage"
   ],
   "summary": "SciCode decomposes 80 scientist-authored research coding problems into 338 subproblems; a main problem counts solved only when every subproblem and the full integration are correct."
  },
  {
   "aliases": [],
   "canonical_id": "scienceqa",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "scienceqa",
   "models_covered": 0,
   "name": "ScienceQA",
   "reasons": [],
   "summary": "21,208 multimodal science multiple-choice questions annotated with lectures and explanations for chain-of-thought training; the leaderboard has sat frozen well above the human baseline since early 2024."
  },
  {
   "aliases": [
    "Scientific Press Release"
   ],
   "canonical_id": "scientific_press_release",
   "category": "generation",
   "disposition": "unassessed",
   "id": "scientific_press_release",
   "models_covered": 0,
   "name": "Scientific Press Release (BIG-bench)",
   "reasons": [],
   "summary": "A 50-item BIG-bench generation task: rewrite a physics paper headline as a press-release title, scored with BLEU."
  },
  {
   "aliases": [
    "SciEval: A Multi-Level Large Language Model Evaluation Benchmark for Scientific Research"
   ],
   "canonical_id": "scieval",
   "category": "domain",
   "disposition": "unassessed",
   "id": "scieval",
   "models_covered": 0,
   "name": "SciEval",
   "reasons": [],
   "summary": "About 18,000 objective and subjective questions in chemistry, physics and biology, scored on basic knowledge, application, calculation and research ability."
  },
  {
   "aliases": [],
   "canonical_id": "sciknoweval",
   "category": "domain",
   "disposition": "unassessed",
   "id": "sciknoweval",
   "models_covered": 0,
   "name": "SciKnowEval",
   "reasons": [],
   "summary": "A five-level scientific-knowledge benchmark, from memorization to real-world application, across biology, chemistry, physics and materials science, with two incompatible dataset versions in current use."
  },
  {
   "aliases": [
    "SciQ dataset",
    "Crowdsourcing Multiple Choice Science Questions"
   ],
   "canonical_id": "sciq",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "sciq",
   "models_covered": 0,
   "name": "SciQ",
   "reasons": [],
   "summary": "13,679 crowdsourced 4-way science questions; lm-eval scores log-likelihood accuracy on the 1,000-item test split after prepending the support paragraph."
  },
  {
   "aliases": [
    "SciReasoner eval",
    "SciReason"
   ],
   "canonical_id": "scireasoner",
   "category": "domain",
   "disposition": "unassessed",
   "id": "scireasoner",
   "models_covered": 0,
   "name": "SciReasoner",
   "reasons": [],
   "summary": "OpenCompass suite of scientific sequence and text tasks from the SciReasoner paper, spanning molecules, proteins, materials, and DNA/RNA."
  },
  {
   "aliases": [
    "SciReasoner1.5",
    "SciReasoner1_5",
    "scireasoner1_5"
   ],
   "canonical_id": "scireasoner1_5",
   "category": "domain",
   "disposition": "unassessed",
   "id": "scireasoner1_5",
   "models_covered": 0,
   "name": "SciReasoner 1.5 (OpenCompass)",
   "reasons": [],
   "summary": "OpenCompass local SciReasoner 1.5 slice: OQMD and JARVIS-DFT material regression plus GO biological-process, TM-score, and DUD-E pair tasks."
  },
  {
   "aliases": [
    "SCORE",
    "Systematic COnsistency and Robustness Evaluation",
    "score_robustness"
   ],
   "canonical_id": "score",
   "category": "composite",
   "disposition": "unassessed",
   "id": "score",
   "models_covered": 0,
   "name": "SCORE (Systematic COnsistency and Robustness Evaluation)",
   "reasons": [],
   "summary": "NVIDIA's non-adversarial robustness suite: re-run MMLU-Pro, AGIEval MCQ, and MATH under prompt, choice-order, and seed changes, reporting accuracy plus consistency rate."
  },
  {
   "aliases": [
    "ScreenSpot Pro"
   ],
   "canonical_id": "screenspot_pro",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "screenspot_pro",
   "models_covered": 2,
   "name": "ScreenSpot-Pro",
   "reasons": [],
   "summary": "1,581 instructions testing whether a model can point to the right UI element in authentic, high-resolution screenshots of 23 professional applications."
  },
  {
   "aliases": [
    "ScreenSpot-Pro agentic",
    "ScreenSpot-Pro zoom-in"
   ],
   "canonical_id": "screenspot_pro_tools",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "screenspot_pro_tools",
   "models_covered": 2,
   "name": "ScreenSpot-Pro (with tools)",
   "reasons": [],
   "summary": "ScreenSpot-Pro scores produced by an iterative search, crop or zoom strategy instead of a single-shot prediction over the full screenshot."
  },
  {
   "aliases": [
    "SCROLLS",
    "Standardized CompaRison Over Long Language Sequences",
    "tau/scrolls"
   ],
   "canonical_id": "scrolls",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "scrolls",
   "models_covered": 0,
   "name": "SCROLLS (Standardized CompaRison Over Long Language Sequences)",
   "reasons": [],
   "summary": "Tel Aviv University suite of seven long-text English tasks (summarisation, QA, NLI) that require synthesising information across naturally long documents."
  },
  {
   "aliases": [
    "SE-Bench: Benchmarking Self-Evolution with Knowledge Internalization"
   ],
   "canonical_id": "se_bench",
   "category": "coding",
   "disposition": "unassessed",
   "id": "se_bench",
   "models_covered": 0,
   "name": "SE-Bench",
   "reasons": [],
   "summary": "A diagnostic coding test that renames NumPy into a fake library so a score reflects whether an agent internalized new APIs, not old knowledge or hard reasoning."
  },
  {
   "aliases": [
    "SE-Eval"
   ],
   "canonical_id": "se_eval",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "se_eval",
   "models_covered": 0,
   "name": "SE-Eval",
   "reasons": [],
   "summary": "Human-rated speech-editing set of 9,151 clips with 1-5 MOS and consistency labels, released as ground truth for scoring speech editors while the ICME 2026 paper stays anonymous."
  },
  {
   "aliases": [
    "SEA-HELM",
    "Southeast Asian Holistic Evaluation of Language Models",
    "BHASA"
   ],
   "canonical_id": "seahelm",
   "category": "composite",
   "disposition": "unassessed",
   "id": "seahelm",
   "models_covered": 0,
   "name": "SEA-HELM (Southeast Asian Holistic Evaluation of Language Models)",
   "reasons": [],
   "summary": "AI Singapore's Southeast Asian LLM suite across NLP classics, LLM-specifics, linguistics, culture and safety, covering Filipino, Indonesian, Tamil, Thai and Vietnamese."
  },
  {
   "aliases": [
    "sec_qa",
    "SecQA v1",
    "SecQA v2"
   ],
   "canonical_id": "sec_qa",
   "category": "domain",
   "disposition": "unassessed",
   "id": "sec_qa",
   "models_covered": 0,
   "name": "SecQA",
   "reasons": [],
   "summary": "242 GPT-4-generated multiple-choice questions on computer security, in two difficulty tiers (v1: 127, v2: 115), scored on accuracy."
  },
  {
   "aliases": [
    "SeedBench: A Multi-task Benchmark for Evaluating Large Language Models in Seed Science"
   ],
   "canonical_id": "seedbench",
   "category": "domain",
   "disposition": "unassessed",
   "id": "seedbench",
   "models_covered": 0,
   "name": "SeedBench",
   "reasons": [],
   "summary": "2,264 expert-validated questions across 11 task types simulating gene retrieval, gene-function analysis and variety breeding for rice, scored by accuracy, macro-F1 or ROUGE-L per task."
  },
  {
   "aliases": [
    "Self-awareness"
   ],
   "canonical_id": "self_awareness",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "self_awareness",
   "models_covered": 0,
   "name": "self_awareness (BIG-bench)",
   "reasons": [],
   "summary": "Programmatic BIG-bench task with eight subtasks that score self-identification, capability claims, and code or environment inspection."
  },
  {
   "aliases": [
    "Self Evaluation Courtroom"
   ],
   "canonical_id": "self_evaluation_courtroom",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "self_evaluation_courtroom",
   "models_covered": 0,
   "name": "self_evaluation_courtroom (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench self-play courtroom: four copies of one model argue, judge, and rate 29 authored cases on a 1\u20139 scale."
  },
  {
   "aliases": [
    "Self Evaluation of Tutoring"
   ],
   "canonical_id": "self_evaluation_tutoring",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "self_evaluation_tutoring",
   "models_covered": 0,
   "name": "self_evaluation_tutoring (BIG-bench Self Evaluation of Tutoring)",
   "reasons": [],
   "summary": "BIG-bench self-play tutoring: one model teaches another on seven topics, then a third copy rates the tutor from 1 to 9."
  },
  {
   "aliases": [
    "Self-Instruct",
    "self-instruct",
    "user_oriented_instructions"
   ],
   "canonical_id": "self_instruct",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "self_instruct",
   "models_covered": 0,
   "name": "Self Instruct (HELM)",
   "reasons": [],
   "summary": "HELM Instruct scenario that scores 252 expert-written Self-Instruct tasks with a 1-5 Helpfulness critique, not the 52k generated training set."
  },
  {
   "aliases": [
    "SParC",
    "semantic parsing in context"
   ],
   "canonical_id": "semantic_parsing_in_context_sparc",
   "category": "coding",
   "disposition": "unassessed",
   "id": "semantic_parsing_in_context_sparc",
   "models_covered": 0,
   "name": "semantic_parsing_in_context_sparc (BIG-bench SParC)",
   "reasons": [],
   "summary": "BIG-bench copy of the SParC development split: 1,203 context-dependent text-to-SQL items scored with BLEU, not execution accuracy."
  },
  {
   "aliases": [
    "Spider (BIG-bench)",
    "BIG-bench Spider"
   ],
   "canonical_id": "semantic_parsing_spider",
   "category": "coding",
   "disposition": "unassessed",
   "id": "semantic_parsing_spider",
   "models_covered": 0,
   "name": "semantic_parsing_spider (BIG-bench Spider)",
   "reasons": [],
   "summary": "BIG-bench wrap of Spider's 1,034-item development set: English question plus schema to SQL, scored by BLEU rather than official Spider exact match."
  },
  {
   "aliases": [],
   "canonical_id": "sentence_ambiguity",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "sentence_ambiguity",
   "models_covered": 0,
   "name": "Sentence Ambiguity",
   "reasons": [],
   "summary": "A 60-item BIG-bench true/false task that labels author-written English sentences laced with hedges, fragments, or partial truths."
  },
  {
   "aliases": [
    "SEvenLLM",
    "SEVENLLM",
    "SEvenLLM-Bench",
    "SEVENLLM-Dataset"
   ],
   "canonical_id": "sevenllm",
   "category": "domain",
   "disposition": "unassessed",
   "id": "sevenllm",
   "models_covered": 0,
   "name": "SEvenLLM (SEvenLLM-Bench)",
   "reasons": [],
   "summary": "Bilingual English/Chinese cyber-threat-intelligence bench of 1,300 test items: 100 four-way MCQs and 1,200 free-form QA items on incident analysis."
  },
  {
   "aliases": [
    "shc_bmt_med",
    "BMT-Status"
   ],
   "canonical_id": "shc_bmt",
   "category": "domain",
   "disposition": "unassessed",
   "id": "shc_bmt",
   "models_covered": 0,
   "name": "BMT-Status",
   "reasons": [],
   "summary": "Private Stanford Health Care MedHELM task: given a hematology note, answer yes or no on whether the patient later received a bone marrow or stem-cell transplant."
  },
  {
   "aliases": [
    "shc_cdi_med",
    "CDI-QA"
   ],
   "canonical_id": "shc_cdi",
   "category": "domain",
   "disposition": "unassessed",
   "id": "shc_cdi",
   "models_covered": 0,
   "name": "CDI-QA",
   "reasons": [],
   "summary": "Private Stanford Health Care MedHELM task: verify from a hospital note whether a documented clinical condition is supported, answering A for yes or B for no."
  },
  {
   "aliases": [
    "shc_conf_med",
    "MedConfInfo"
   ],
   "canonical_id": "shc_conf",
   "category": "domain",
   "disposition": "unassessed",
   "id": "shc_conf",
   "models_covered": 0,
   "name": "MedConfInfo",
   "reasons": [],
   "summary": "Private MedHELM task from Stanford adolescent notes: say whether a visit note contains sensitive content that should be withheld from a parent portal view."
  },
  {
   "aliases": [
    "shc_ent_med",
    "ENT-Referral"
   ],
   "canonical_id": "shc_ent",
   "category": "domain",
   "disposition": "unassessed",
   "id": "shc_ent",
   "models_covered": 0,
   "name": "ENT-Referral",
   "reasons": [],
   "summary": "Private Stanford Health Care MedHELM task: from a clinical note, answer whether an ENT referral is supported, with A yes, B no, or C no mention."
  },
  {
   "aliases": [
    "shc_gip_med",
    "HospiceReferral"
   ],
   "canonical_id": "shc_gip",
   "category": "domain",
   "disposition": "unassessed",
   "id": "shc_gip",
   "models_covered": 0,
   "name": "HospiceReferral",
   "reasons": [],
   "summary": "Private Stanford Health Care MedHELM task: from a palliative care note, answer yes or no on whether the patient is eligible for hospice referral."
  },
  {
   "aliases": [
    "shc_privacy_med",
    "PrivacyDetection"
   ],
   "canonical_id": "shc_privacy",
   "category": "domain",
   "disposition": "unassessed",
   "id": "shc_privacy",
   "models_covered": 0,
   "name": "PrivacyDetection",
   "reasons": [],
   "summary": "Private MedHELM task: decide whether a patient-portal message contains confidential or privacy-leaking information, answering A for yes or B for no."
  },
  {
   "aliases": [
    "shc_proxy_med",
    "ProxySender"
   ],
   "canonical_id": "shc_proxy",
   "category": "domain",
   "disposition": "unassessed",
   "id": "shc_proxy",
   "models_covered": 0,
   "name": "ProxySender",
   "reasons": [],
   "summary": "Private MedHELM task: decide whether a clinician-facing portal message was sent by the patient or by a proxy such as a parent or spouse."
  },
  {
   "aliases": [
    "shc_ptbm_med",
    "PTBM",
    "parent training in behavior management"
   ],
   "canonical_id": "shc_ptbm",
   "category": "domain",
   "disposition": "unassessed",
   "id": "shc_ptbm",
   "models_covered": 0,
   "name": "ADHD-Behavior",
   "reasons": [],
   "summary": "Private MedHELM binary task: given a pediatric ADHD visit note, say whether the clinician recommended parent training in behavior management."
  },
  {
   "aliases": [
    "shc_sei_med",
    "SEI",
    "side effects inquiry"
   ],
   "canonical_id": "shc_sei",
   "category": "domain",
   "disposition": "unassessed",
   "id": "shc_sei",
   "models_covered": 0,
   "name": "ADHD-MedEffects",
   "reasons": [],
   "summary": "Private MedHELM binary task: given a pediatric ADHD medication-follow-up note, say whether it documents side-effect inquiry."
  },
  {
   "aliases": [
    "shc_sequoia_med",
    "Sequoia clinic referral"
   ],
   "canonical_id": "shc_sequoia",
   "category": "domain",
   "disposition": "unassessed",
   "id": "shc_sequoia",
   "models_covered": 0,
   "name": "ClinicReferral",
   "reasons": [],
   "summary": "Private MedHELM binary task: from a palliative-care note, decide whether a patient is eligible for referral to Stanford's Sequoia clinic."
  },
  {
   "aliases": [
    "Similarities Test for Abstraction"
   ],
   "canonical_id": "similarities_abstraction",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "similarities_abstraction",
   "models_covered": 0,
   "name": "Similarities Test for Abstraction",
   "reasons": [],
   "summary": "A 76-item BIG-bench similarities test that asks how two objects are alike, scoring the abstract option over concrete distractors."
  },
  {
   "aliases": [
    "simp_turing_concept",
    "Simplicity priors for Turing-complete concept learning"
   ],
   "canonical_id": "simp_turing_concept",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "simp_turing_concept",
   "models_covered": 0,
   "name": "Alignment of Simplicity Priors for Turing-Complete Concept Learning",
   "reasons": [],
   "summary": "A 6,390-query BIG-bench task that tests few-shot learning of 426 P3 binary-string programs from machine-teaching witness sets."
  },
  {
   "aliases": [
    "Simple arithmetic"
   ],
   "canonical_id": "simple_arithmetic",
   "category": "math",
   "disposition": "unassessed",
   "id": "simple_arithmetic",
   "models_covered": 0,
   "name": "simple_arithmetic (BIG-bench programmatic addition template)",
   "reasons": [],
   "summary": "A BIG-bench Python template that asks for n-digit integer sums, tagged as an example task rather than a capability ranking."
  },
  {
   "aliases": [
    "Simple arithmetic example using JSON"
   ],
   "canonical_id": "simple_arithmetic_json",
   "category": "math",
   "disposition": "unassessed",
   "id": "simple_arithmetic_json",
   "models_covered": 0,
   "name": "simple_arithmetic_json (BIG-bench JSON arithmetic template)",
   "reasons": [],
   "summary": "A 30-item BIG-bench JSON template of one- to three-digit addition, written as an example for task authors rather than as a capability benchmark."
  },
  {
   "aliases": [
    "Simple multiple choice arithmetic example using JSON"
   ],
   "canonical_id": "simple_arithmetic_json_multiple_choice",
   "category": "math",
   "disposition": "unassessed",
   "id": "simple_arithmetic_json_multiple_choice",
   "models_covered": 0,
   "name": "simple_arithmetic_json_multiple_choice (BIG-bench JSON MC arithmetic template)",
   "reasons": [],
   "summary": "An 8-item BIG-bench JSON template of one-digit addition with four numeric choices, written as an example of the multiple-choice JSON format."
  },
  {
   "aliases": [
    "Simple arithmetic example using JSON with subtasks"
   ],
   "canonical_id": "simple_arithmetic_json_subtasks",
   "category": "math",
   "disposition": "unassessed",
   "id": "simple_arithmetic_json_subtasks",
   "models_covered": 0,
   "name": "simple_arithmetic_json_subtasks (BIG-bench nested JSON arithmetic template)",
   "reasons": [],
   "summary": "A 30-item BIG-bench JSON template that splits one- to three-digit addition into three digit-length subtasks, as an example of nested JSON tasks."
  },
  {
   "aliases": [],
   "canonical_id": "simple_arithmetic_multiple_targets_json",
   "category": "math",
   "disposition": "unassessed",
   "id": "simple_arithmetic_multiple_targets_json",
   "models_covered": 0,
   "name": "simple_arithmetic_multiple_targets_json (BIG-bench multi-target JSON template)",
   "reasons": [],
   "summary": "A 10-item BIG-bench JSON template of one-digit addition that accepts either a numeral, an English word, or both as a correct target."
  },
  {
   "aliases": [
    "simple-cooccurrence-bias",
    "GPT-3 occupation gender association test"
   ],
   "canonical_id": "simple_cooccurrence_bias",
   "category": "safety",
   "disposition": "unassessed",
   "id": "simple_cooccurrence_bias",
   "models_covered": 0,
   "name": "Simple Cooccurrence Bias",
   "reasons": [],
   "summary": "A next-token association test: after 'The {occupation} was a', compare likelihoods of male versus female gender identifiers."
  },
  {
   "aliases": [
    "Alignment Questionnaire"
   ],
   "canonical_id": "simple_ethical_questions",
   "category": "safety",
   "disposition": "unassessed",
   "id": "simple_ethical_questions",
   "models_covered": 0,
   "name": "Simple Ethical Questions (BIG-bench)",
   "reasons": [],
   "summary": "A 115-item BIG-bench questionnaire of science-fiction and social dilemmas with one preferred answer among four options."
  },
  {
   "aliases": [
    "SST",
    "Simple Safety Tests",
    "simple_safety_tests"
   ],
   "canonical_id": "simple_safety_tests",
   "category": "safety",
   "disposition": "unassessed",
   "id": "simple_safety_tests",
   "models_covered": 0,
   "name": "SimpleSafetyTests",
   "reasons": [],
   "summary": "A 100-prompt English suite of requests that models should refuse, covering five high-severity harm areas."
  },
  {
   "aliases": [
    "simple text editing"
   ],
   "canonical_id": "simple_text_editing",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "simple_text_editing",
   "models_covered": 0,
   "name": "Simple Text Editing (BIG-bench)",
   "reasons": [],
   "summary": "A 47-item BIG-bench task: apply a stated edit to a short Alan Turing paragraph and return the full edited text."
  },
  {
   "aliases": [],
   "canonical_id": "simpleqa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "simpleqa",
   "models_covered": 0,
   "name": "SimpleQA",
   "reasons": [],
   "summary": "4,326 short, adversarially-collected fact-seeking questions with one indisputable answer, graded CORRECT/INCORRECT/NOT_ATTEMPTED by a model; GPT-4o scored under 40% at release."
  },
  {
   "aliases": [
    "SocialIQA",
    "Social Interaction QA"
   ],
   "canonical_id": "siqa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "siqa",
   "models_covered": 0,
   "name": "Social IQa",
   "reasons": [],
   "summary": "35,364 public three-way multiple-choice questions about people's motivations and reactions in social situations; a 2019 AI2/UW benchmark with a dead leaderboard and no current model-card coverage."
  },
  {
   "aliases": [],
   "canonical_id": "skillsbench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "skillsbench",
   "models_covered": 0,
   "name": "SkillsBench",
   "reasons": [],
   "summary": "SkillsBench evaluates how well agent skills work and how effectively agents use them across practical tasks."
  },
  {
   "aliases": [
    "SLM-Bench",
    "Small Language Model-Benchmark"
   ],
   "canonical_id": "slm_bench",
   "category": "composite",
   "disposition": "unassessed",
   "id": "slm_bench",
   "models_covered": 0,
   "name": "SLM-Bench",
   "reasons": [],
   "summary": "Fine-tuning benchmark of 15 sub-7B-class models on 23 NLP datasets with 11 correctness, runtime, cost, energy, and CO2 metrics on four hardware setups."
  },
  {
   "aliases": [
    "SLR-Bench",
    "Scalable Logical Reasoning Benchmark",
    "slr_bench_group"
   ],
   "canonical_id": "slr_bench_group",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "slr_bench_group",
   "models_covered": 0,
   "name": "SLR-Bench (group)",
   "reasons": [],
   "summary": "Over 19,000 auto-synthesised inductive-logic-programming tasks across 20 curriculum levels, symbolically verified; the lm-evaluation-harness group averages verifiable-reward across five constituent tasks."
  },
  {
   "aliases": [
    "SmolInstruct",
    "SMolInstruct",
    "LlaSMol dataset"
   ],
   "canonical_id": "smolinstruct",
   "category": "domain",
   "disposition": "unassessed",
   "id": "smolinstruct",
   "models_covered": 0,
   "name": "SMolInstruct",
   "reasons": [],
   "summary": "A 14-task small-molecule chemistry instruction set of about 3.3 million samples, used both to train LlaSMol and as an OpenCompass evaluation."
  },
  {
   "aliases": [
    "Snarks",
    "BBH snarks"
   ],
   "canonical_id": "snarks",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "snarks",
   "models_covered": 0,
   "name": "SNARKS (BIG-bench sarcasm contrast set)",
   "reasons": [],
   "summary": "A 181-item BIG-bench contrast set: choose which of two minimally edited English sentences is sarcastic."
  },
  {
   "aliases": [],
   "canonical_id": "social_support",
   "category": "safety",
   "disposition": "unassessed",
   "id": "social_support",
   "models_covered": 0,
   "name": "Social Support",
   "reasons": [],
   "summary": "A BIG-bench task that asks a model to classify a comment from an online support community as supportive, neutral or unsupportive of the post it replies to."
  },
  {
   "aliases": [
    "SoSBench",
    "SOSBench: Benchmarking Safety Alignment on Scientific Knowledge",
    "SoSBench: Benchmarking Safety Alignment on Six Scientific Domains"
   ],
   "canonical_id": "sosbench",
   "category": "safety",
   "disposition": "unassessed",
   "id": "sosbench",
   "models_covered": 0,
   "name": "SOSBench",
   "reasons": [],
   "summary": "3,000 regulation-grounded, hazard-focused prompts across six scientific domains, testing whether models refuse policy-violating requests that require real scientific expertise to recognise."
  },
  {
   "aliases": [
    "SPADE-Bench",
    "Spontaneous Plan-Action Divergence Evaluation"
   ],
   "canonical_id": "spade_bench",
   "category": "safety",
   "disposition": "unassessed",
   "id": "spade_bench",
   "models_covered": 0,
   "name": "SPADE-Bench",
   "reasons": [],
   "summary": "300 paired tool-use scenarios that score whether an agent changes its stated plan under pressure while its action stays the same; the metric is Pass@5 deception rate."
  },
  {
   "aliases": [
    "spanish_bench"
   ],
   "canonical_id": "spanish_bench",
   "category": "composite",
   "disposition": "unassessed",
   "id": "spanish_bench",
   "models_covered": 0,
   "name": "SpanishBench",
   "reasons": [],
   "summary": "Fifteen-task European Spanish suite spanning commonsense, QA, NLI, translation, math, summarisation and more, run as one lm-evaluation-harness group tag."
  },
  {
   "aliases": [],
   "canonical_id": "spelling_bee",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "spelling_bee",
   "models_covered": 0,
   "name": "Spelling Bee",
   "reasons": [],
   "summary": "A BIG-bench task modelled on the New York Times Spelling Bee puzzle: given seven letters, list as many valid English words over four letters as possible, scored by a pangram-weighted point system."
  },
  {
   "aliases": [
    "Spider 1.0"
   ],
   "canonical_id": "spider",
   "category": "coding",
   "disposition": "unassessed",
   "id": "spider",
   "models_covered": 0,
   "name": "Spider",
   "reasons": [],
   "summary": "Spider is a large-scale, cross-domain text-to-SQL benchmark: a model must generate a correct SQL query for a natural-language question against one of 200 databases it has not seen during training."
  },
  {
   "aliases": [],
   "canonical_id": "sports_understanding",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "sports_understanding",
   "models_covered": 0,
   "name": "Sports Understanding",
   "reasons": [],
   "summary": "A BIG-bench task that asks a model to judge whether a made-up sentence pairing an athlete with a sport-specific action is plausible or implausible."
  },
  {
   "aliases": [
    "Stanford Question Answering Dataset",
    "SQuAD 1.1"
   ],
   "canonical_id": "squad",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "squad",
   "models_covered": 0,
   "name": "SQuAD",
   "reasons": [],
   "summary": "SQuAD (2016) and SQuAD 2.0 (2018) are the canonical Wikipedia span-extraction reading sets; the official leaderboard has sat above human performance since 2019, so a modern score there is uninformative."
  },
  {
   "aliases": [
    "SQuAD2.0"
   ],
   "canonical_id": "squad20",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "squad20",
   "models_covered": 0,
   "name": "SQuAD 2.0 (OpenCompass squad20)",
   "reasons": [],
   "summary": "OpenCompass's squad20 task runs the SQuAD 2.0 dev set, prompting a model to extract an answer span or say \"impossible to answer\" for adversarial unanswerable questions."
  },
  {
   "aliases": [
    "based-squad",
    "hazyresearch/based-squad"
   ],
   "canonical_id": "squad_completion",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "squad_completion",
   "models_covered": 0,
   "name": "SQuAD completion (Based / lm-eval)",
   "reasons": [],
   "summary": "Based and lm-eval rewrite 2,984 SQuAD validation items as next-token completions, scored by case-insensitive contains."
  },
  {
   "aliases": [
    "SQuADShifts",
    "SQuAD-Shifts",
    "SquadShifts"
   ],
   "canonical_id": "squad_shifts",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "squad_shifts",
   "models_covered": 0,
   "name": "squad_shifts (BIG-bench SQuADShifts)",
   "reasons": [],
   "summary": "BIG-bench zero-shot span QA on SQuADShifts plus SQuAD 1.1 dev, covering Wikipedia, NYT, Reddit and Amazon reviews."
  },
  {
   "aliases": [
    "SQuAD2"
   ],
   "canonical_id": "squadv2",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "squadv2",
   "models_covered": 0,
   "name": "SQuAD 2.0 (lm-evaluation-harness squadv2)",
   "reasons": [],
   "summary": "lm-evaluation-harness's squadv2 task runs the SQuAD 2.0 validation set, scoring exact match and F1 separately for answerable and unanswerable questions via the official squad_v2 metric."
  },
  {
   "aliases": [
    "SRBench: A Living Benchmark for Symbolic Regression",
    "Symbolic Regression Benchmark"
   ],
   "canonical_id": "srbench",
   "category": "math",
   "disposition": "unassessed",
   "id": "srbench",
   "models_covered": 0,
   "name": "SRBench",
   "reasons": [],
   "summary": "OpenCompass's LLM adaptation of the SRBench symbolic-regression project: given numeric input-output samples, a model must output a closed-form formula, scored by fit quality and symbolic equivalence."
  },
  {
   "aliases": [
    "PatientInstruct"
   ],
   "canonical_id": "starr_patient_instructions",
   "category": "domain",
   "disposition": "unassessed",
   "id": "starr_patient_instructions",
   "models_covered": 0,
   "name": "STARR Patient Instructions (PatientInstruct)",
   "reasons": [],
   "summary": "A MedHELM scenario, built from private Stanford Health Care records, that asks a model to write post-procedure patient instructions from a diagnosis, procedure and clinical notes."
  },
  {
   "aliases": [],
   "canonical_id": "stereoset",
   "category": "safety",
   "disposition": "unassessed",
   "id": "stereoset",
   "models_covered": 0,
   "name": "StereoSet",
   "reasons": [],
   "summary": "Crowdsourced test of whether a language model prefers stereotypical over anti-stereotypical associations across gender, race, religion and profession, while staying fluent."
  },
  {
   "aliases": [
    "StoryCloze",
    "ROCStories Cloze Test"
   ],
   "canonical_id": "storycloze",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "storycloze",
   "models_covered": 0,
   "name": "Story Cloze Test",
   "reasons": [],
   "summary": "Story Cloze Test asks a model to pick the correct one of two endings to a four-sentence story; the original 2016 set has documented annotation biases exploitable without real story understanding."
  },
  {
   "aliases": [],
   "canonical_id": "strange_stories",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "strange_stories",
   "models_covered": 0,
   "name": "Strange Stories",
   "reasons": [],
   "summary": "BIG-bench task built on a clinical theory-of-mind battery that asks a model to infer characters' beliefs, intentions and non-literal meaning from short narratives."
  },
  {
   "aliases": [
    "Did Aristotle Use a Laptop?"
   ],
   "canonical_id": "strategyqa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "strategyqa",
   "models_covered": 0,
   "name": "StrategyQA",
   "reasons": [],
   "summary": "A 2,780-question yes/no benchmark whose reasoning steps are never stated in the question, testing whether a model can infer and chain an implicit strategy to reach the answer."
  },
  {
   "aliases": [],
   "canonical_id": "strong_reject",
   "category": "safety",
   "disposition": "unassessed",
   "id": "strong_reject",
   "models_covered": 0,
   "name": "StrongREJECT",
   "reasons": [],
   "summary": "Measures how much harmful, specific and convincing content a model produces on forbidden prompts, with or without a jailbreak applied, using an LLM-judged rubric."
  },
  {
   "aliases": [],
   "canonical_id": "subject_verb_agreement",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "subject_verb_agreement",
   "models_covered": 0,
   "name": "Subject-Verb Agreement",
   "reasons": [],
   "summary": "Programmatic BIG-bench task that scores whether a model assigns higher probability to the grammatically correct verb form across syntactic constructions with long-distance or nested agreement."
  },
  {
   "aliases": [],
   "canonical_id": "sudoku",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "sudoku",
   "models_covered": 0,
   "name": "Sudoku",
   "reasons": [],
   "summary": "Programmatic BIG-bench task where a model interactively fills in procedurally generated Sudoku puzzles one cell at a time, scored on command syntax, rule-following and full solution."
  },
  {
   "aliases": [],
   "canonical_id": "sufficient_information",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "sufficient_information",
   "models_covered": 0,
   "name": "Sufficient Information",
   "reasons": [],
   "summary": "Tiny BIG-bench task of 39 short word problems that checks whether a model answers when given enough information and says 'I do not know' when it is not."
  },
  {
   "aliases": [
    "suicide_risk (BIG-bench)"
   ],
   "canonical_id": "suicide_risk",
   "category": "safety",
   "disposition": "unassessed",
   "id": "suicide_risk",
   "models_covered": 0,
   "name": "Estimating Risk of Suicide",
   "reasons": [],
   "summary": "A 50-item BIG-bench task asking a model to grade suicide risk in short Reddit-derived posts on a four-level scale against expert labels."
  },
  {
   "aliases": [
    "SummarizationScenario"
   ],
   "canonical_id": "summarization",
   "category": "generation",
   "disposition": "unassessed",
   "id": "summarization",
   "models_covered": 0,
   "name": "HELM Summarization",
   "reasons": [],
   "summary": "HELM's single-document summarization scenario, which scores models on abstracting BBC news articles (XSum) or CNN/DailyMail articles into short summaries, mainly by ROUGE-2."
  },
  {
   "aliases": [
    "SummEdits Benchmark"
   ],
   "canonical_id": "summedits",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "summedits",
   "models_covered": 0,
   "name": "SummEdits",
   "reasons": [],
   "summary": "6,348-item binary benchmark asking whether a small edit to a factually consistent summary preserved consistency with its source document, across 10 text domains."
  },
  {
   "aliases": [
    "SummScreen: A Dataset for Abstractive Screenplay Summarization"
   ],
   "canonical_id": "summscreen",
   "category": "generation",
   "disposition": "unassessed",
   "id": "summscreen",
   "models_covered": 0,
   "name": "SummScreen",
   "reasons": [],
   "summary": "Pairs of TV series transcripts and human-written episode recaps, testing long-document abstractive summarization where plot detail is scattered across dialogue."
  },
  {
   "aliases": [
    "SUMO climate summarization"
   ],
   "canonical_id": "sumosum",
   "category": "domain",
   "disposition": "unassessed",
   "id": "sumosum",
   "models_covered": 0,
   "name": "SUMOSum (HELM climate-claims summarization)",
   "reasons": [],
   "summary": "HELM's repurposing of the climate subset of the SUMO web-claims dataset as document-to-title summarization, distinct from the original paper's claim-verification task."
  },
  {
   "aliases": [
    "Super General Language Understanding Evaluation benchmark",
    "SGLUE"
   ],
   "canonical_id": "super_glue",
   "category": "composite",
   "disposition": "unassessed",
   "id": "super_glue",
   "models_covered": 0,
   "name": "SuperGLUE (Super General Language Understanding Evaluation benchmark)",
   "reasons": [],
   "summary": "An eight-task English language-understanding suite that replaced GLUE once models exceeded its human baseline, itself since superseded by harder generative and agentic benchmarks."
  },
  {
   "aliases": [
    "SuperCLUE Agent"
   ],
   "canonical_id": "superclue_agent",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "superclue_agent",
   "models_covered": 0,
   "name": "SuperCLUE-Agent",
   "reasons": [],
   "summary": "Chinese-native agent eval of tool use, planning and memory across ten tasks; GPT-4 led the 2023 table at 80.56, with no published item count or licence."
  },
  {
   "aliases": [
    "SC-Safety",
    "SuperCLUE Safety"
   ],
   "canonical_id": "superclue_safety",
   "category": "safety",
   "disposition": "unassessed",
   "id": "superclue_safety",
   "models_covered": 0,
   "name": "SuperCLUE-Safety",
   "reasons": [],
   "summary": "Chinese multi-turn adversarial safety eval of 4,912 open-ended items scored 0-2 across traditional safety, responsible AI and instruction attacks."
  },
  {
   "aliases": [
    "AX-b",
    "AXb",
    "SuperGLUE_AX_b",
    "AX_b"
   ],
   "canonical_id": "superglue_ax_b",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "superglue_ax_b",
   "models_covered": 0,
   "name": "SuperGLUE AX-b (Broad Coverage Diagnostics)",
   "reasons": [],
   "summary": "SuperGLUE diagnostic: 1,104 English sentence pairs recast from the GLUE diagnostic as two-way entailment, scored with Matthews correlation."
  },
  {
   "aliases": [
    "AX-g",
    "AXg",
    "SuperGLUE_AX_g",
    "AX_g",
    "Winogender (SuperGLUE)"
   ],
   "canonical_id": "superglue_ax_g",
   "category": "safety",
   "disposition": "unassessed",
   "id": "superglue_ax_g",
   "models_covered": 0,
   "name": "SuperGLUE AX-g (Winogender Schema Diagnostics)",
   "reasons": [],
   "summary": "SuperGLUE Winogender diagnostic: 356 English premise\u2013hypothesis pairs that test whether pronoun gender flips an entailment decision."
  },
  {
   "aliases": [
    "CB",
    "CommitmentBank",
    "SuperGLUE_CB"
   ],
   "canonical_id": "superglue_cb",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "superglue_cb",
   "models_covered": 0,
   "name": "SuperGLUE CB (CommitmentBank)",
   "reasons": [],
   "summary": "SuperGLUE's three-class entailment recast of CommitmentBank: 250/56/250 English pairs, scored with accuracy and macro-F1."
  },
  {
   "aliases": [
    "COPA",
    "Choice of Plausible Alternatives",
    "SuperGLUE_COPA"
   ],
   "canonical_id": "superglue_copa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "superglue_copa",
   "models_covered": 0,
   "name": "SuperGLUE COPA (Choice of Plausible Alternatives)",
   "reasons": [],
   "summary": "SuperGLUE packaging of COPA: pick the more plausible cause or effect of a one-sentence English premise from two alternatives."
  },
  {
   "aliases": [
    "MultiRC",
    "SuperGLUE_MultiRC",
    "Multi-Sentence Reading Comprehension"
   ],
   "canonical_id": "superglue_multirc",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "superglue_multirc",
   "models_covered": 0,
   "name": "SuperGLUE MultiRC (Multi-Sentence Reading Comprehension)",
   "reasons": [],
   "summary": "SuperGLUE MultiRC: decide which candidate answers to a question are true given a paragraph that requires more than one sentence."
  },
  {
   "aliases": [
    "ReCoRD",
    "SuperGLUE_ReCoRD",
    "record"
   ],
   "canonical_id": "superglue_record",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "superglue_record",
   "models_covered": 0,
   "name": "SuperGLUE ReCoRD (Reading Comprehension with Commonsense Reasoning Dataset)",
   "reasons": [],
   "summary": "SuperGLUE's cloze reading-comprehension task: recover a masked entity in a CNN/Daily Mail query, scored by token-level F1 and exact match."
  },
  {
   "aliases": [
    "SuperGLUE_RTE",
    "sglue_rte"
   ],
   "canonical_id": "superglue_rte",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "superglue_rte",
   "models_covered": 0,
   "name": "SuperGLUE RTE (Recognizing Textual Entailment)",
   "reasons": [],
   "summary": "SuperGLUE's two-way textual-entailment task, reused from GLUE RTE: decide whether a hypothesis is entailed by a premise, scored by accuracy."
  },
  {
   "aliases": [
    "WiC",
    "SuperGLUE_WiC",
    "wic"
   ],
   "canonical_id": "superglue_wic",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "superglue_wic",
   "models_covered": 0,
   "name": "SuperGLUE WiC (Word-in-Context)",
   "reasons": [],
   "summary": "SuperGLUE's word-sense task: decide whether a polysemous word has the same sense in two short sentences, scored by accuracy."
  },
  {
   "aliases": [
    "SuperGLUE_WSC",
    "wsc",
    "wsc.fixed"
   ],
   "canonical_id": "superglue_wsc",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "superglue_wsc",
   "models_covered": 0,
   "name": "SuperGLUE WSC (Winograd Schema Challenge, SuperGLUE recast)",
   "reasons": [],
   "summary": "SuperGLUE's recast Winograd Schema Challenge: decide whether a marked pronoun refers to a marked noun in one English sentence, scored by accuracy."
  },
  {
   "aliases": [
    "Super-GPQA",
    "Super GPQA"
   ],
   "canonical_id": "supergpqa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "supergpqa",
   "models_covered": 0,
   "name": "SuperGPQA",
   "reasons": [],
   "summary": "Graduate-level multiple-choice questions across 13 disciplines, 72 fields and 285 subfields, filtered with a human-LLM pipeline to drop trivial and ambiguous items."
  },
  {
   "aliases": [
    "Simple Variations on Arithmetic Math word Problems"
   ],
   "canonical_id": "svamp",
   "category": "math",
   "disposition": "unassessed",
   "id": "svamp",
   "models_covered": 0,
   "name": "SVAMP",
   "reasons": [],
   "summary": "A 1,000-item challenge set of grade-school arithmetic word problems made by applying question, reasoning and structure variations to existing MWPs."
  },
  {
   "aliases": [
    "Situations With Adversarial Generations"
   ],
   "canonical_id": "swag",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "swag",
   "models_covered": 0,
   "name": "SWAG (Situations With Adversarial Generations)",
   "reasons": [],
   "summary": "113k four-way multiple-choice questions asking which of four captions plausibly continues a video-derived situation; HellaSwag's direct predecessor."
  },
  {
   "aliases": [
    "Swahili-English Paremiologic Competence"
   ],
   "canonical_id": "swahili_english_proverbs",
   "category": "translation",
   "disposition": "unassessed",
   "id": "swahili_english_proverbs",
   "models_covered": 0,
   "name": "Swahili-English Proverbs",
   "reasons": [],
   "summary": "BIG-bench multiple-choice task matching a Kiswahili proverb to its closest English-language equivalent among four options."
  },
  {
   "aliases": [
    "Structured Web Data Extraction"
   ],
   "canonical_id": "swde",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "swde",
   "models_covered": 0,
   "name": "SWDE (lm-evaluation-harness zero-shot extraction task)",
   "reasons": [],
   "summary": "Zero-shot task: given a full movie webpage's text in-context, extract the value for a named attribute such as release date or director."
  },
  {
   "aliases": [],
   "canonical_id": "swe_agent",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "swe_agent",
   "models_covered": 0,
   "name": "SWE-agent",
   "reasons": [],
   "summary": "SWE-agent is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "swe_agi",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_agi",
   "models_covered": 0,
   "name": "SWE-AGI",
   "reasons": [],
   "summary": "SWE-AGI asks an agent to build a production-scale program, such as a parser or SAT solver, from a written specification in the little-known MoonBit language."
  },
  {
   "aliases": [
    "SWE-bench Full",
    "SWE-bench Original"
   ],
   "canonical_id": "swe_bench",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench",
   "models_covered": 0,
   "name": "SWE-bench",
   "reasons": [],
   "summary": "SWE-bench tests whether a model can resolve real GitHub issues by generating a patch, checked by the repository's own test suite; Verified now supersedes the original set for most reporting."
  },
  {
   "aliases": [],
   "canonical_id": "swe_bench_agent",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_agent",
   "models_covered": 61,
   "name": "SWE-bench Agent",
   "reasons": [],
   "summary": "An internal model-card key for an agentic (not single-shot patch) SWE-bench score; no publisher, paper or dataset for it under this name was found."
  },
  {
   "aliases": [
    "SWE-bench-CL"
   ],
   "canonical_id": "swe_bench_cl",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_cl",
   "models_covered": 0,
   "name": "SWE-Bench-CL",
   "reasons": [],
   "summary": "A 273-task continual-learning reformulation of SWE-bench Verified: eight chronological repository sequences with forgetting and transfer metrics."
  },
  {
   "aliases": [],
   "canonical_id": "swe_bench_dummy_test_dataset",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "swe_bench_dummy_test_dataset",
   "models_covered": 0,
   "name": "swe-bench-dummy-test-dataset",
   "reasons": [],
   "summary": "swe-bench-dummy-test-dataset is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "swe_bench_extra",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_extra",
   "models_covered": 0,
   "name": "SWE-bench Extra",
   "reasons": [],
   "summary": "SWE-bench Extra is a 6,415-instance dataset of real GitHub issue-and-fix pairs, built with the SWE-bench methodology to extend beyond the original benchmark's repositories."
  },
  {
   "aliases": [
    "SWE-bench-java-verified",
    "SWE-bench Java"
   ],
   "canonical_id": "swe_bench_java",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_java",
   "models_covered": 0,
   "name": "SWE-bench-java",
   "reasons": [],
   "summary": "A 91-instance, human-screened Java SWE-bench set from six popular repositories, scored by fail-to-pass unit tests."
  },
  {
   "aliases": [
    "SWE-bench_Lite",
    "SWE-Bench-Lite"
   ],
   "canonical_id": "swe_bench_lite",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_lite",
   "models_covered": 0,
   "name": "SWE-bench Lite",
   "reasons": [],
   "summary": "A 300-task, single-file-edit subset of original SWE-bench, kept as a cheaper Python issue-resolution reporting split."
  },
  {
   "aliases": [
    "SWE-bench Live",
    "SWE-bench Goes Live"
   ],
   "canonical_id": "swe_bench_live",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_live",
   "models_covered": 0,
   "name": "SWE-bench-Live",
   "reasons": [],
   "summary": "A continuously updated SWE-bench-style issue-resolution set built from recent GitHub issues, with frozen Lite/Verified splits and a growing full split."
  },
  {
   "aliases": [
    "SWE-bench_Multilingual"
   ],
   "canonical_id": "swe_bench_multilingual",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_multilingual",
   "models_covered": 2,
   "name": "SWE-bench Multilingual",
   "reasons": [],
   "summary": "SWE-bench Multilingual extends SWE-bench's real-GitHub-issue patch task to 300 tasks across 9 non-Python languages and 42 repositories."
  },
  {
   "aliases": [
    "SWE-bench M",
    "SWE-bench MM"
   ],
   "canonical_id": "swe_bench_multimodal",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_multimodal",
   "models_covered": 2,
   "name": "SWE-bench Multimodal",
   "reasons": [],
   "summary": "SWE-bench Multimodal tests GitHub-issue patching on JavaScript/TypeScript repositories where the issue includes an image, such as a bug screenshot or design mockup."
  },
  {
   "aliases": [],
   "canonical_id": "swe_bench_mutated",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_mutated",
   "models_covered": 0,
   "name": "SWE-Bench-Mutated",
   "reasons": [],
   "summary": "SWE-Bench-Mutated is a tool and methodology that rewrites formal SWE-bench issue text into realistic chat-style developer queries, to test agents on more realistic inputs."
  },
  {
   "aliases": [
    "SWE-Bench Pro"
   ],
   "canonical_id": "swe_bench_pro",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_pro",
   "models_covered": 5,
   "name": "SWE-bench Pro",
   "reasons": [],
   "summary": "Scale AI's harder, contamination-resistant successor in spirit to SWE-bench Verified: 1,865 long-horizon tasks across public copyleft, held-out and private commercial codebases."
  },
  {
   "aliases": [
    "SWE-Bench-ProMax",
    "SWE-bench ProMax"
   ],
   "canonical_id": "swe_bench_promax",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_promax",
   "models_covered": 0,
   "name": "SWE-Bench ProMax",
   "reasons": [],
   "summary": "170 expert-curated multilingual refactoring tasks from post-2025 commits; gold patches average 11.4 files and 261.6 lines."
  },
  {
   "aliases": [
    "SWE-bench-Science"
   ],
   "canonical_id": "swe_bench_science",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_science",
   "models_covered": 0,
   "name": "SWE-bench Science",
   "reasons": [],
   "summary": "119 repository-level scientific coding tasks across 98 GitHub projects and 20 domains, scored by held-out programmatic verifiers."
  },
  {
   "aliases": [
    "SWE-bench-V"
   ],
   "canonical_id": "swe_bench_verified",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_bench_verified",
   "models_covered": 120,
   "name": "SWE-bench Verified",
   "reasons": [],
   "summary": "A 500-task, human-screened subset of SWE-bench that OpenAI released with the SWE-bench authors to remove unfair or impossible samples; now the default SWE-bench reference."
  },
  {
   "aliases": [],
   "canonical_id": "swe_bench_verified_o1_reasoning_high_results",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "swe_bench_verified_o1_reasoning_high_results",
   "models_covered": 0,
   "name": "SWE-Bench-Verified-O1-reasoning-high-results",
   "reasons": [],
   "summary": "SWE-Bench-Verified-O1-reasoning-high-results is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "swe_evo",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_evo",
   "models_covered": 0,
   "name": "SWE-EVO",
   "reasons": [],
   "summary": "SWE-EVO gives coding agents a real project release note and asks for the multi-file changes it describes, checked against the project's own tests."
  },
  {
   "aliases": [
    "SWE-Explore-Bench"
   ],
   "canonical_id": "swe_explore",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_explore",
   "models_covered": 0,
   "name": "SWE-Explore",
   "reasons": [],
   "summary": "SWE-Explore isolates repository exploration from patch generation, scoring the ranked code regions an agent returns for a GitHub issue under a fixed line budget."
  },
  {
   "aliases": [],
   "canonical_id": "swe_gym",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_gym",
   "models_covered": 0,
   "name": "SWE-Gym",
   "reasons": [],
   "summary": "SWE-Gym is a training environment of real Python issue-resolution tasks; agents trained on it are typically reported by resolve rate on SWE-bench Verified and Lite, not a separate SWE-Gym leaderboard."
  },
  {
   "aliases": [
    "SWE-Lancer Diamond"
   ],
   "canonical_id": "swe_lancer",
   "category": "coding",
   "disposition": "unassessed",
   "id": "swe_lancer",
   "models_covered": 0,
   "name": "SWE-Lancer",
   "reasons": [],
   "summary": "OpenAI's benchmark of over 1,400 real Upwork freelance tasks on the Expensify codebase, worth $1 million in actual historical payouts, scored in dollars earned rather than percent resolved."
  },
  {
   "aliases": [],
   "canonical_id": "swe_nfi",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "swe_nfi",
   "models_covered": 0,
   "name": "SWE-NFI",
   "reasons": [],
   "summary": "SWE-NFI is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "swe_polybench",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "swe_polybench",
   "models_covered": 0,
   "name": "SWE-PolyBench",
   "reasons": [],
   "summary": "SWE-PolyBench is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "swe_rebench",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "swe_rebench",
   "models_covered": 0,
   "name": "SWE-rebench",
   "reasons": [],
   "summary": "SWE-rebench is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "swe_together",
   "category": "knowledge",
   "disposition": "unverified",
   "id": "swe_together",
   "models_covered": 0,
   "name": "SWE-Together",
   "reasons": [
    "model glm-5.2 is absent from reference set for domain agentic",
    "model gpt-5.5 is absent from reference set for domain agentic",
    "no qualifying current frontier/open coverage from different organizations",
    "review is not approved",
    "usefulness is not established",
    "usefulness is unknown"
   ],
   "summary": "SWE-Together is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "swe_touch",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "swe_touch",
   "models_covered": 0,
   "name": "SWE-Touch",
   "reasons": [],
   "summary": "SWE-Touch is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "swedish_to_german_proverbs",
   "category": "translation",
   "disposition": "unassessed",
   "id": "swedish_to_german_proverbs",
   "models_covered": 0,
   "name": "Swedish to German Proverbs",
   "reasons": [],
   "summary": "BIG-bench multiple-choice task: pick the German proverb closest in meaning to a given Swedish proverb, from four options."
  },
  {
   "aliases": [
    "Swiss-Bench SBP-003",
    "SBP-003"
   ],
   "canonical_id": "swiss_bench_003",
   "category": "composite",
   "disposition": "unassessed",
   "id": "swiss_bench_003",
   "models_covered": 0,
   "name": "Swiss-Bench 003",
   "reasons": [],
   "summary": "808 Swiss-adapted items that extend HAAS with self-graded D7 reliability and judged D8 security scores aimed at FINMA and nDSG deployment questions."
  },
  {
   "aliases": [
    "Swiss-Bench 002",
    "SBP-002"
   ],
   "canonical_id": "swiss_bench_sbp_002",
   "category": "domain",
   "disposition": "unassessed",
   "id": "swiss_bench_sbp_002",
   "models_covered": 0,
   "name": "Swiss-Bench SBP-002",
   "reasons": [],
   "summary": "Trilingual Swiss regulatory-compliance eval of 395 expert items; a three-judge panel found even the top model only 38.2% correct under zero retrieval."
  },
  {
   "aliases": [],
   "canonical_id": "sycophancy",
   "category": "safety",
   "disposition": "unassessed",
   "id": "sycophancy",
   "models_covered": 0,
   "name": "Sycophancy Eval (inspect_evals, 'Are you sure?')",
   "reasons": [],
   "summary": "Asks a model factual questions, then challenges a correct answer with 'Are you sure?' to see if it sticks to the truth or capitulates."
  },
  {
   "aliases": [
    "SIT"
   ],
   "canonical_id": "symbol_interpretation",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "symbol_interpretation",
   "models_covered": 0,
   "name": "Symbol Interpretation",
   "reasons": [],
   "summary": "BIG-bench Lite task: pick which sentence correctly describes a 'structure' \u2014 a sequence of six emoji pieces \u2014 across five adversarial variants."
  },
  {
   "aliases": [
    "HELM synthetic efficiency"
   ],
   "canonical_id": "synthetic_efficiency",
   "category": "generation",
   "disposition": "unassessed",
   "id": "synthetic_efficiency",
   "models_covered": 0,
   "name": "Synthetic efficiency (HELM)",
   "reasons": [],
   "summary": "HELM runtime probe: generate from fixed public-domain prompts while varying prompt length, output length and tokenizer."
  },
  {
   "aliases": [
    "HELM synthetic reasoning",
    "synthetic_reasoning (symbolic)",
    "Synthetic reasoning (abstract symbols)"
   ],
   "canonical_id": "synthetic_reasoning",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "synthetic_reasoning",
   "models_covered": 0,
   "name": "Synthetic reasoning (HELM, abstract symbols)",
   "reasons": [],
   "summary": "HELM LIME-style synthetic tasks: match a pattern, substitute variables, or induce a rule over abstract symbols."
  },
  {
   "aliases": [
    "SRN"
   ],
   "canonical_id": "synthetic_reasoning_natural",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "synthetic_reasoning_natural",
   "models_covered": 0,
   "name": "Synthetic Reasoning (Natural Language)",
   "reasons": [],
   "summary": "HELM scenario: given natural-language conditional rules and facts, deduce the correct consequent, at easy/medium/hard abstraction levels."
  },
  {
   "aliases": [
    "Tabular Math Word Problems",
    "PromptPG"
   ],
   "canonical_id": "tabmwp",
   "category": "math",
   "disposition": "unassessed",
   "id": "tabmwp",
   "models_covered": 0,
   "name": "TabMWP",
   "reasons": [],
   "summary": "38,431 grade-level math word problems that require reasoning over both a short question and an accompanying table, mixing free-text and multiple-choice answers."
  },
  {
   "aliases": [],
   "canonical_id": "taboo",
   "category": "generation",
   "disposition": "unassessed",
   "id": "taboo",
   "models_covered": 0,
   "name": "Taboo",
   "reasons": [],
   "summary": "BIG-bench self-play game: one model instance must define a target word without using forbidden related words, and a second instance must guess it."
  },
  {
   "aliases": [
    "Travel Agent Compassion",
    "inspect_evals/tac",
    "tac_welfare"
   ],
   "canonical_id": "tac",
   "category": "safety",
   "disposition": "unassessed",
   "id": "tac",
   "models_covered": 0,
   "name": "TAC (Travel Agent Compassion)",
   "reasons": [],
   "summary": "Thirteen travel-booking scenarios (52 after augmentation) where a tool-using agent must avoid animal-exploitation tickets the user never named."
  },
  {
   "aliases": [
    "Topics in Algorithmic COde generation",
    "BAAI/TACO",
    "FlagOpen/TACO"
   ],
   "canonical_id": "taco",
   "category": "coding",
   "disposition": "unassessed",
   "id": "taco",
   "models_covered": 0,
   "name": "TACO (Topics in Algorithmic COde generation)",
   "reasons": [],
   "summary": "BAAI TACO: 1,000 competition-style Python problems (plus a 25k-problem train split) scored by executing generated code as pass@k."
  },
  {
   "aliases": [
    "TalkDown",
    "talkdown condescension"
   ],
   "canonical_id": "talkdown",
   "category": "safety",
   "disposition": "unassessed",
   "id": "talkdown",
   "models_covered": 0,
   "name": "TalkDown (BIG-bench)",
   "reasons": [],
   "summary": "A 652-item BIG-bench programmatic task: say whether an English Reddit utterance is condescending."
  },
  {
   "aliases": [
    "tau2-bench",
    "tau^2-bench",
    "tau 2",
    "inspect_evals/tau2"
   ],
   "canonical_id": "tau2",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "tau2",
   "models_covered": 0,
   "name": "\u03c4\u00b2-bench",
   "reasons": [],
   "summary": "Sierra \u03c4\u00b2-bench: a tool-using support agent plus a simulated user, including telecom where the user can call tools too."
  },
  {
   "aliases": [
    "tau-bench",
    "tau bench"
   ],
   "canonical_id": "tau_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "tau_bench",
   "models_covered": 54,
   "name": "\u03c4-bench",
   "reasons": [],
   "summary": "Scores whether a tool-using agent can hold a policy-following conversation with a simulated customer and leave the backend in the right state."
  },
  {
   "aliases": [
    "TellMeWhy",
    "tell me why"
   ],
   "canonical_id": "tellmewhy",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "tellmewhy",
   "models_covered": 0,
   "name": "TellMeWhy (BIG-bench)",
   "reasons": [],
   "summary": "A 3,562-item BIG-bench task: answer a why-question about a five-sentence English story in free text."
  },
  {
   "aliases": [
    "Temporal Sequences",
    "BBH temporal_sequences"
   ],
   "canonical_id": "temporal_sequences",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "temporal_sequences",
   "models_covered": 0,
   "name": "Temporal Sequences (BIG-bench)",
   "reasons": [],
   "summary": "A 1,000-item BIG-bench task: pick the only free hour range for an untimed event; BBH keeps 250 items."
  },
  {
   "aliases": [
    "Verb Tense",
    "BIG-bench tense"
   ],
   "canonical_id": "tense",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "tense",
   "models_covered": 0,
   "name": "Verb Tense (BIG-bench)",
   "reasons": [],
   "summary": "A 286-item BIG-bench task: rewrite an English sentence into a named tense without changing the rest of the meaning."
  },
  {
   "aliases": [
    "Terminal-Bench 1.0",
    "Terminal-Bench-Core",
    "T-Bench"
   ],
   "canonical_id": "terminal_bench",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "terminal_bench",
   "models_covered": 53,
   "name": "Terminal-Bench",
   "reasons": [],
   "summary": "Terminal-Bench 1.0 measures whether an AI agent can complete real command-line tasks in a sandboxed Docker terminal, graded by automated tests; superseded by later major versions."
  },
  {
   "aliases": [
    "Terminal-Bench 2",
    "TB2",
    "Terminal-Bench 2.1"
   ],
   "canonical_id": "terminal_bench_2",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "terminal_bench_2",
   "models_covered": 29,
   "name": "Terminal-Bench 2.0",
   "reasons": [],
   "summary": "A harder, more heavily verified 89-task remake of Terminal-Bench, released with the Harbor evaluation package; Terminal-Bench 2.1 later patched 28 of its tasks."
  },
  {
   "aliases": [],
   "canonical_id": "terminal_bench_2_0",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "terminal_bench_2_0",
   "models_covered": 0,
   "name": "Terminal-Bench 2.0",
   "reasons": [],
   "summary": "Terminal-Bench 2.0 evaluates agents completing software, science and system tasks in Docker environments."
  },
  {
   "aliases": [],
   "canonical_id": "terminal_bench_2_verified",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "terminal_bench_2_verified",
   "models_covered": 0,
   "name": "Terminal-Bench 2.0 Verified",
   "reasons": [],
   "summary": "This reviewed derivative fixes environment and instruction issues in Terminal-Bench 2.0 while preserving task logic where stated."
  },
  {
   "aliases": [],
   "canonical_id": "terminal_bench_3_0",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "terminal_bench_3_0",
   "models_covered": 0,
   "name": "Terminal-Bench 3.0",
   "reasons": [],
   "summary": "Terminal-Bench 3.0 is the v3.0.0 Harbor dataset release with task instructions, environments, tests and solutions."
  },
  {
   "aliases": [],
   "canonical_id": "terminal_bench_lilt",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "terminal_bench_lilt",
   "models_covered": 0,
   "name": "Terminal-Bench-LILT",
   "reasons": [],
   "summary": "300 authentic coding tasks in ten languages test agents on multilingual software-development problems."
  },
  {
   "aliases": [],
   "canonical_id": "terminal_bench_science",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "terminal_bench_science",
   "models_covered": 0,
   "name": "Terminal-Bench-Science",
   "reasons": [],
   "summary": "Terminal-Bench-Science evaluates real computational research tasks across life, physical, earth, mathematical and engineering sciences."
  },
  {
   "aliases": [],
   "canonical_id": "terminal_bench_v2_1",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "terminal_bench_v2_1",
   "models_covered": 0,
   "name": "Terminal-Bench v2.1",
   "reasons": [],
   "summary": "The supplied v2.1 lead resolves only to an Artificial Analysis logo asset, not an authoritative benchmark release or task registry."
  },
  {
   "aliases": [],
   "canonical_id": "terminal_bench_v4_0",
   "category": "agentic",
   "disposition": "unverified",
   "id": "terminal_bench_v4_0",
   "models_covered": 0,
   "name": "Terminal-Bench v4.0",
   "reasons": [
    "no qualifying current frontier/open coverage from different organizations",
    "review is not approved",
    "usefulness is not established",
    "usefulness is unknown"
   ],
   "summary": "The supplied v4.0 lead resolves only to an Artificial Analysis logo asset, not an authoritative benchmark release or task registry."
  },
  {
   "aliases": [
    "TEval",
    "Tool-Eval"
   ],
   "canonical_id": "teval",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "teval",
   "models_covered": 0,
   "name": "T-Eval",
   "reasons": [],
   "summary": "A bilingual tool-use benchmark that scores instruction, planning, reasoning, retrieval, understanding and review as separate steps rather than one success bit."
  },
  {
   "aliases": [],
   "canonical_id": "text_navigation_game",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "text_navigation_game",
   "models_covered": 0,
   "name": "Text Navigation Game",
   "reasons": [],
   "summary": "BIG-bench Text Navigation Game has a model issue free-text moves across turns to reach a target on an ASCII-grid maze."
  },
  {
   "aliases": [
    "Thai Exam",
    "thai-exam"
   ],
   "canonical_id": "thai_exam",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "thai_exam",
   "models_covered": 0,
   "name": "ThaiExam",
   "reasons": [],
   "summary": "A suite of Thai multiple-choice exams (ONET, IC, TGAT, TPAT-1, A-Level) used to test Thai knowledge in Typhoon and later as a HELM scenario."
  },
  {
   "aliases": [
    "FACTS Grounding",
    "FACTS Grounding v1"
   ],
   "canonical_id": "the_facts_grounding_leaderboard",
   "category": "generation",
   "disposition": "unassessed",
   "id": "the_facts_grounding_leaderboard",
   "models_covered": 0,
   "name": "The FACTS Grounding Leaderboard",
   "reasons": [],
   "summary": "Google FACTS Grounding scores whether long-form answers stay faithful to a supplied document, using an ensemble of LLM judges plus an eligibility filter."
  },
  {
   "aliases": [
    "FACTS Leaderboard Suite",
    "FACTS Score"
   ],
   "canonical_id": "the_facts_leaderboard",
   "category": "composite",
   "disposition": "unassessed",
   "id": "the_facts_leaderboard",
   "models_covered": 0,
   "name": "The FACTS Leaderboard",
   "reasons": [],
   "summary": "Google FACTS suite averages multimodal, parametric, search and grounding-v2 tracks into one FACTS Score on public and private splits."
  },
  {
   "aliases": [],
   "canonical_id": "the_pile",
   "category": "generation",
   "disposition": "unassessed",
   "id": "the_pile",
   "models_covered": 0,
   "name": "The Pile",
   "reasons": [],
   "summary": "HELM evaluates language-model perplexity on a test slice of The Pile across its component domains."
  },
  {
   "aliases": [],
   "canonical_id": "theagentcompany",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "theagentcompany",
   "models_covered": 0,
   "name": "TheAgentCompany",
   "reasons": [],
   "summary": "TheAgentCompany has an agent complete 175 long-horizon professional tasks in a simulated software company to measure real-work automation."
  },
  {
   "aliases": [
    "Theorem QA"
   ],
   "canonical_id": "theoremqa",
   "category": "math",
   "disposition": "unassessed",
   "id": "theoremqa",
   "models_covered": 0,
   "name": "TheoremQA",
   "reasons": [],
   "summary": "800 university-level STEM questions that require applying a named theorem, with answers as numbers, lists, booleans or multiple-choice letters."
  },
  {
   "aliases": [],
   "canonical_id": "threecb",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "threecb",
   "models_covered": 0,
   "name": "ThreeCB",
   "reasons": [],
   "summary": "3CB (Catastrophic Cyber Capabilities Benchmark) scores LLM agents on capture-the-flag cyber offense challenges mapped to MITRE ATT&CK techniques."
  },
  {
   "aliases": [],
   "canonical_id": "timedial",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "timedial",
   "models_covered": 0,
   "name": "TimeDial",
   "reasons": [],
   "summary": "BIG-bench TimeDial asks models to select the correct answer for masked temporal spans in dialogue context."
  },
  {
   "aliases": [
    "tiny Benchmarks"
   ],
   "canonical_id": "tinybenchmarks",
   "category": "composite",
   "disposition": "unassessed",
   "id": "tinybenchmarks",
   "models_covered": 0,
   "name": "tinyBenchmarks",
   "reasons": [],
   "summary": "One-hundred-item IRT-selected subsets of Open LLM Leaderboard tasks that reconstruct full-benchmark accuracy from a few percent of the original items."
  },
  {
   "aliases": [
    "TMMLU Plus"
   ],
   "canonical_id": "tmmluplus",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "tmmluplus",
   "models_covered": 0,
   "name": "TMMLU+",
   "reasons": [],
   "summary": "TMMLU+ is an lm-evaluation-harness task family for Taiwanese Mandarin knowledge questions."
  },
  {
   "aliases": [],
   "canonical_id": "topical_chat",
   "category": "generation",
   "disposition": "unassessed",
   "id": "topical_chat",
   "models_covered": 0,
   "name": "Topical-Chat",
   "reasons": [],
   "summary": "BIG-bench Topical-Chat evaluates open-domain response generation in conversations grounded in topical information."
  },
  {
   "aliases": [],
   "canonical_id": "toxigen",
   "category": "safety",
   "disposition": "unassessed",
   "id": "toxigen",
   "models_covered": 71,
   "name": "ToxiGen",
   "reasons": [],
   "summary": "A machine-generated dataset of 274,000 toxic and benign statements about 13 minority groups, used to test whether models can catch subtle, implicit hate speech."
  },
  {
   "aliases": [],
   "canonical_id": "tracking_shuffled_objects",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "tracking_shuffled_objects",
   "models_covered": 0,
   "name": "Tracking Shuffled Objects",
   "reasons": [],
   "summary": "BIG-bench task for tracking object positions through a sequence of swaps."
  },
  {
   "aliases": [],
   "canonical_id": "training_on_test_set",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "training_on_test_set",
   "models_covered": 0,
   "name": "Training on Test Set",
   "reasons": [],
   "summary": "BIG-bench task designed to detect evidence that a language model was trained on benchmark data."
  },
  {
   "aliases": [],
   "canonical_id": "translation",
   "category": "translation",
   "disposition": "unassessed",
   "id": "translation",
   "models_covered": 0,
   "name": "Translation Tasks",
   "reasons": [],
   "summary": "A family of translation tasks configured in lm-evaluation-harness."
  },
  {
   "aliases": [],
   "canonical_id": "triviaqa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "triviaqa",
   "models_covered": 0,
   "name": "TriviaQA",
   "reasons": [],
   "summary": "Trivia questions with answer-alias lists, run two very different ways -- extractive reading comprehension over given evidence, or open-domain closed-book recall -- with very different scores."
  },
  {
   "aliases": [],
   "canonical_id": "triviaqarc",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "triviaqarc",
   "models_covered": 0,
   "name": "TriviaQA RC",
   "reasons": [],
   "summary": "TriviaQA reading comprehension evaluated through the OpenCompass TriviaQArc configuration."
  },
  {
   "aliases": [],
   "canonical_id": "truthfulqa",
   "category": "safety",
   "disposition": "unassessed",
   "id": "truthfulqa",
   "models_covered": 50,
   "name": "TruthfulQA",
   "reasons": [],
   "summary": "817 questions written to bait a model into repeating common human misconceptions, testing truthfulness rather than raw knowledge."
  },
  {
   "aliases": [
    "truthfulqa-multi",
    "Multilingual TruthfulQA",
    "Truth Knows No Language"
   ],
   "canonical_id": "truthfulqa_multi",
   "category": "safety",
   "disposition": "unassessed",
   "id": "truthfulqa_multi",
   "models_covered": 0,
   "name": "TruthfulQA-Multi",
   "reasons": [],
   "summary": "Professionally translated TruthfulQA in Basque, Catalan, Galician and Spanish, plus English, scored with MC2 and generation metrics."
  },
  {
   "aliases": [
    "TurBLiMP"
   ],
   "canonical_id": "turblimp_core",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "turblimp_core",
   "models_covered": 0,
   "name": "TurBLiMP Core",
   "reasons": [],
   "summary": "TurBLiMP's core group of 16 Turkish grammaticality phenomena, 1,000 minimal pairs each, run as a single lm-evaluation-harness task."
  },
  {
   "aliases": [
    "Turkish MMLU"
   ],
   "canonical_id": "turkishmmlu",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "turkishmmlu",
   "models_covered": 0,
   "name": "TurkishMMLU",
   "reasons": [],
   "summary": "TurkishMMLU evaluates Turkish-language multiple-choice knowledge across nine high-school subjects."
  },
  {
   "aliases": [
    "tweetSentBR"
   ],
   "canonical_id": "tweetsentbr",
   "category": "domain",
   "disposition": "unassessed",
   "id": "tweetsentbr",
   "models_covered": 0,
   "name": "TweetSentBR",
   "reasons": [],
   "summary": "TweetSentBR evaluates sentiment classification for Brazilian Portuguese tweets."
  },
  {
   "aliases": [
    "BIG-bench Twenty Questions"
   ],
   "canonical_id": "twenty_questions",
   "category": "agentic",
   "disposition": "unassessed",
   "id": "twenty_questions",
   "models_covered": 0,
   "name": "Twenty Questions",
   "reasons": [],
   "summary": "Two model instances play Twenty Questions, communicating a hidden concept through yes-or-no answers."
  },
  {
   "aliases": [
    "Twitter African-American English",
    "Twitter AAE corpus"
   ],
   "canonical_id": "twitter_aae",
   "category": "domain",
   "disposition": "unassessed",
   "id": "twitter_aae",
   "models_covered": 0,
   "name": "TwitterAAE",
   "reasons": [],
   "summary": "HELM evaluates language-model bits per byte on 50,000 AAE-aligned and 50,000 White-aligned tweets."
  },
  {
   "aliases": [
    "TyDiQA",
    "TyDiQA-GoldP"
   ],
   "canonical_id": "tydiqa",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "tydiqa",
   "models_covered": 0,
   "name": "TyDi QA",
   "reasons": [],
   "summary": "204,000 information-seeking question-answer pairs across 11 typologically diverse languages, collected without translation, testing extractive QA beyond English."
  },
  {
   "aliases": [
    "UCCB",
    "Ugandan Cultural and Cognitive Benchmark"
   ],
   "canonical_id": "uccb",
   "category": "domain",
   "disposition": "unassessed",
   "id": "uccb",
   "models_covered": 0,
   "name": "Uganda Cultural and Cognitive Benchmark",
   "reasons": [],
   "summary": "UCCB contains 1,039 open-ended questions across 24 Ugandan cultural domains, scored by model-based grading."
  },
  {
   "aliases": [
    "ulqa_",
    "uleval",
    "ULUT",
    "CELEP1",
    "CELEP2",
    "lambada_uyghur"
   ],
   "canonical_id": "ulqa",
   "category": "composite",
   "disposition": "unassessed",
   "id": "ulqa",
   "models_covered": 0,
   "name": "ULQA (Uyghur language eval group)",
   "reasons": [],
   "summary": "lm-eval group of Uyghur textbook, exam, and cloze tasks covering basic language, literature, and last-word prediction."
  },
  {
   "aliases": [
    "UncheatableEval",
    "UE"
   ],
   "canonical_id": "uncheatable_eval",
   "category": "generation",
   "disposition": "unassessed",
   "id": "uncheatable_eval",
   "models_covered": 0,
   "name": "Uncheatable Eval",
   "reasons": [],
   "summary": "Rolling perplexity on monthly snapshots of new Wikipedia, GitHub, BBC, arXiv, bioRxiv and AO3 text, meant to limit train-set leakage."
  },
  {
   "aliases": [
    "BIG-bench Understanding Fables"
   ],
   "canonical_id": "understanding_fables",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "understanding_fables",
   "models_covered": 0,
   "name": "Understanding Fables",
   "reasons": [],
   "summary": "189 paraphrased fables paired with five candidate morals test narrative understanding and cross-domain generalization."
  },
  {
   "aliases": [
    "BIG-bench Undo Permutation",
    "Reordering"
   ],
   "canonical_id": "undo_permutation",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "undo_permutation",
   "models_covered": 0,
   "name": "Reordering",
   "reasons": [],
   "summary": "Three synthetic subtasks ask models to reconstruct ordered sentences from scrambled words, characters or swaps."
  },
  {
   "aliases": [
    "Unit Conversion"
   ],
   "canonical_id": "unit_conversion",
   "category": "math",
   "disposition": "unassessed",
   "id": "unit_conversion",
   "models_covered": 0,
   "name": "unit_conversion (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench unit-conversion suite: eight subtasks, 24,000 items, mixing multiple-choice conversions and 10% relative-error numeric answers."
  },
  {
   "aliases": [
    "Unit Interpretation"
   ],
   "canonical_id": "unit_interpretation",
   "category": "math",
   "disposition": "unassessed",
   "id": "unit_interpretation",
   "models_covered": 0,
   "name": "unit_interpretation (BIG-bench)",
   "reasons": [],
   "summary": "BIG-bench 100-item, 5-way multiple-choice probe of implicit unit arithmetic in short English word problems."
  },
  {
   "aliases": [
    "BIG-bench Unnatural In-Context Learning"
   ],
   "canonical_id": "unnatural_in_context_learning",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "unnatural_in_context_learning",
   "models_covered": 0,
   "name": "Unnatural In-Context Learning",
   "reasons": [],
   "summary": "Synthetic identity, date, reversal and arithmetic subtasks test in-context pattern induction outside natural training distributions."
  },
  {
   "aliases": [
    "UNQOVERing Stereotyping Biases via Underspecified Questions"
   ],
   "canonical_id": "unqover",
   "category": "safety",
   "disposition": "unassessed",
   "id": "unqover",
   "models_covered": 0,
   "name": "UnQover",
   "reasons": [],
   "summary": "UnQover probes gender, nationality, ethnicity and religion stereotypes with underspecified span-question-answering templates."
  },
  {
   "aliases": [
    "Word Scrambling and Manipulation Tasks"
   ],
   "canonical_id": "unscramble",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "unscramble",
   "models_covered": 0,
   "name": "Unscramble",
   "reasons": [],
   "summary": "A battery of 5 character-manipulation tasks from the GPT-3 paper that asks a model to recover an original word from a scrambled, cycled, reversed, or noise-inserted version of it."
  },
  {
   "aliases": [
    "USACOBench"
   ],
   "canonical_id": "usaco",
   "category": "coding",
   "disposition": "unassessed",
   "id": "usaco",
   "models_covered": 0,
   "name": "USACO",
   "reasons": [],
   "summary": "A 307-problem benchmark built from USA Computing Olympiad contests that tests whether a model can write a Python program that passes hidden stdin/stdout test cases under time and memory limits."
  },
  {
   "aliases": [
    "MathArena USAMO 2026"
   ],
   "canonical_id": "usamo_2026",
   "category": "math",
   "disposition": "unassessed",
   "id": "usamo_2026",
   "models_covered": 5,
   "name": "USAMO 2026",
   "reasons": [],
   "summary": "Grades full written proofs, not just final answers, for the six 2026 USA Mathematical Olympiad problems."
  },
  {
   "aliases": [
    "Visual Fidelity Against Text-bias"
   ],
   "canonical_id": "v_fat",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "v_fat",
   "models_covered": 0,
   "name": "V-FAT",
   "reasons": [],
   "summary": "V-FAT tests whether multimodal models answer from the image when corpus priors or misleading prompts conflict with what is shown."
  },
  {
   "aliases": [
    "Verified Financial LLM Reasoning Benchmark"
   ],
   "canonical_id": "v_fillm",
   "category": "domain",
   "disposition": "unassessed",
   "id": "v_fillm",
   "models_covered": 0,
   "name": "V-FiLLM",
   "reasons": [],
   "summary": "V-FiLLM builds financial table-reasoning questions from executable computation trees so answers are correct by construction and difficulty is controllable."
  },
  {
   "aliases": [],
   "canonical_id": "verifiability_judgment",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "verifiability_judgment",
   "models_covered": 0,
   "name": "Verifiability Judgment",
   "reasons": [],
   "summary": "A HELM scenario that gives a model a generated statement and its cited source and asks it to judge whether the source fully, partially, or does not support the statement."
  },
  {
   "aliases": [],
   "canonical_id": "vga_bench",
   "category": "generation",
   "disposition": "unassessed",
   "id": "vga_bench",
   "models_covered": 0,
   "name": "VGA-Bench",
   "reasons": [],
   "summary": "VGA-Bench scores text-to-video models on aesthetic quality, aesthetic tags, and generation quality using 1,016 prompts and 52 sub-dimensions."
  },
  {
   "aliases": [
    "VGA-Bench V2"
   ],
   "canonical_id": "vga_benchv2",
   "category": "generation",
   "disposition": "unassessed",
   "id": "vga_benchv2",
   "models_covered": 0,
   "name": "VGA-BenchV2",
   "reasons": [],
   "summary": "VGA-BenchV2 keeps VGA-Bench's 52-dimension prompt suite and adds larger human labels, hybrid judges, and aesthetic reward-model fine-tuning."
  },
  {
   "aliases": [],
   "canonical_id": "vibe",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "vibe",
   "models_covered": 0,
   "name": "VIBE",
   "reasons": [],
   "summary": "VIBE is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "vibe_bench",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "vibe_bench",
   "models_covered": 0,
   "name": "VIBE-Bench",
   "reasons": [],
   "summary": "VIBE-Bench tests personalized language models under profile-preference conceptual misalignment using personas and dialogues."
  },
  {
   "aliases": [],
   "canonical_id": "vibe_eval",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "vibe_eval",
   "models_covered": 0,
   "name": "Vibe-Eval",
   "reasons": [],
   "summary": "Vibe-Eval is an open benchmark of 269 visual-understanding prompts for evaluating multimodal chat models."
  },
  {
   "aliases": [
    "Vicuna-80"
   ],
   "canonical_id": "vicuna",
   "category": "instruction-following",
   "disposition": "unassessed",
   "id": "vicuna",
   "models_covered": 0,
   "name": "Vicuna Questions",
   "reasons": [],
   "summary": "A HELM scenario built from the 80 open-ended questions LMSYS used to evaluate Vicuna in 2023, scored in HELM by human critique ratings rather than the original GPT-4-judge protocol."
  },
  {
   "aliases": [
    "vimgolf_single_turn",
    "VimGolf"
   ],
   "canonical_id": "vimgolf_challenges",
   "category": "coding",
   "disposition": "unassessed",
   "id": "vimgolf_challenges",
   "models_covered": 0,
   "name": "VimGolf Challenges (inspect_evals)",
   "reasons": [],
   "summary": "Single-turn inspect_evals task: emit a Vim keystroke string that turns each of 612 public VimGolf inputs into the target buffer."
  },
  {
   "aliases": [],
   "canonical_id": "vitaminc_fact_verification",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "vitaminc_fact_verification",
   "models_covered": 0,
   "name": "VitaminC Fact Verification",
   "reasons": [],
   "summary": "A BIG-bench task built from the VitaminC dataset that asks a model to judge whether a Wikipedia passage supports, refutes, or gives no information about a claim."
  },
  {
   "aliases": [],
   "canonical_id": "vqa_rad",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "vqa_rad",
   "models_covered": 0,
   "name": "VQA-RAD",
   "reasons": [],
   "summary": "The first manually constructed radiology visual question answering dataset, pairing clinician questions about head, chest and abdominal images with reference answers."
  },
  {
   "aliases": [
    "VStar_Bench",
    "vstar-bench",
    "V-Star Bench"
   ],
   "canonical_id": "vstar_bench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "vstar_bench",
   "models_covered": 0,
   "name": "V*Bench",
   "reasons": [],
   "summary": "A 191-question visual-question-answering test built from high-resolution, visually crowded images, designed so a model cannot answer correctly without precisely locating a small target detail."
  },
  {
   "aliases": [],
   "canonical_id": "web_of_lies",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "web_of_lies",
   "models_covered": 0,
   "name": "Web of Lies",
   "reasons": [],
   "summary": "A BIG-bench task that phrases a chain of nested boolean functions as a word problem about people who tell the truth or lie, and asks the model to answer yes or no."
  },
  {
   "aliases": [],
   "canonical_id": "webqs",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "webqs",
   "models_covered": 0,
   "name": "WebQuestions",
   "reasons": [],
   "summary": "WebQuestions evaluates short-answer questions whose answers are grounded in Freebase entities."
  },
  {
   "aliases": [],
   "canonical_id": "what_is_the_tao",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "what_is_the_tao",
   "models_covered": 0,
   "name": "What Is the Tao?",
   "reasons": [],
   "summary": "BIG-bench What Is the Tao compares stylistic elements of translations of a philosophical text."
  },
  {
   "aliases": [
    "opencompass/WikiBench",
    "wikibench-wiki-single_choice_cn"
   ],
   "canonical_id": "wikibench",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "wikibench",
   "models_covered": 0,
   "name": "WikiBench (OpenCompass)",
   "reasons": [],
   "summary": "OpenCompass Chinese four-way Wikipedia-style MCQ set, usually reported under circular evaluation rather than single-pass accuracy."
  },
  {
   "aliases": [
    "WikiText-2",
    "WikiText-103"
   ],
   "canonical_id": "wikitext",
   "category": "generation",
   "disposition": "unassessed",
   "id": "wikitext",
   "models_covered": 0,
   "name": "WikiText",
   "reasons": [],
   "summary": "WikiText scores raw language-modelling quality by perplexity on curated Wikipedia articles; it measures how well a model predicts text, not whether it answers a task correctly."
  },
  {
   "aliases": [
    "WildBench v2"
   ],
   "canonical_id": "wildbench",
   "category": "human-preference",
   "disposition": "unassessed",
   "id": "wildbench",
   "models_covered": 45,
   "name": "WildBench",
   "reasons": [],
   "summary": "1,024 hard tasks mined from over a million real chatbot conversations, scored automatically by an LLM judge against a task-specific checklist."
  },
  {
   "aliases": [],
   "canonical_id": "winogrande",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "winogrande",
   "models_covered": 60,
   "name": "WinoGrande",
   "reasons": [],
   "summary": "A 44k-problem, adversarially filtered successor to the Winograd Schema Challenge, testing commonsense pronoun resolution at scale."
  },
  {
   "aliases": [],
   "canonical_id": "winogrande_afr",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "winogrande_afr",
   "models_covered": 0,
   "name": "Winogrande African Languages",
   "reasons": [],
   "summary": "HELM's winogrande_afr scenario evaluates commonsense pronoun resolution using Winogrande items translated into 11 low-resource African languages."
  },
  {
   "aliases": [],
   "canonical_id": "winowhy",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "winowhy",
   "models_covered": 0,
   "name": "WinoWhy",
   "reasons": [],
   "summary": "WinoWhy tests whether a model can pick the correct justification for a Winograd Schema Challenge answer, not just the answer itself."
  },
  {
   "aliases": [
    "Weapons of Mass Destruction Proxy Benchmark"
   ],
   "canonical_id": "wmdp",
   "category": "safety",
   "disposition": "unassessed",
   "id": "wmdp",
   "models_covered": 0,
   "name": "WMDP (Weapons of Mass Destruction Proxy)",
   "reasons": [],
   "summary": "A 3,668-question multiple-choice proxy for hazardous biosecurity, cybersecurity and chemical-security knowledge, built as a target for unlearning research; a lower score is the safety-desirable outcome."
  },
  {
   "aliases": [],
   "canonical_id": "wmt2016",
   "category": "translation",
   "disposition": "unassessed",
   "id": "wmt2016",
   "models_covered": 0,
   "name": "WMT 2016 (Romanian-English, T5 prompt)",
   "reasons": [],
   "summary": "The lm-evaluation-harness wmt2016 task scores Romanian-to-English translation from the WMT16 test set using a T5-style prompt."
  },
  {
   "aliases": [],
   "canonical_id": "wmt_14",
   "category": "translation",
   "disposition": "unassessed",
   "id": "wmt_14",
   "models_covered": 0,
   "name": "WMT 14",
   "reasons": [],
   "summary": "HELM's WMT_14 scenario scores machine translation on five WMT14 English language pairs using sentence-level BLEU-4."
  },
  {
   "aliases": [],
   "canonical_id": "word_problems_on_sets_and_graphs",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "word_problems_on_sets_and_graphs",
   "models_covered": 0,
   "name": "Word Problems on Sets and Graphs",
   "reasons": [],
   "summary": "A BIG-bench free-response task that checks whether a model tracks set membership and graph paths across three small synthetic puzzle types."
  },
  {
   "aliases": [
    "Sorting Words"
   ],
   "canonical_id": "word_sorting",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "word_sorting",
   "models_covered": 0,
   "name": "Word Sorting (BIG-bench)",
   "reasons": [],
   "summary": "A 1,900-item BIG-bench free-response task: sort English words alphabetically. BIG-bench Hard keeps a 250-item slice."
  },
  {
   "aliases": [
    "word unscrambling"
   ],
   "canonical_id": "word_unscrambling",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "word_unscrambling",
   "models_covered": 0,
   "name": "Word Unscrambling (BIG-bench)",
   "reasons": [],
   "summary": "An 8,917-item BIG-bench free-response task: restore a scrambled letter string to an English word, scored by exact string match."
  },
  {
   "aliases": [],
   "canonical_id": "worldsense",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "worldsense",
   "models_covered": 0,
   "name": "WorldSense",
   "reasons": [],
   "summary": "WorldSense is a synthetic benchmark that tests whether a model can maintain a consistent world model while controlling for dataset bias."
  },
  {
   "aliases": [],
   "canonical_id": "writingbench",
   "category": "generation",
   "disposition": "unassessed",
   "id": "writingbench",
   "models_covered": 0,
   "name": "WritingBench",
   "reasons": [],
   "summary": "1,000 real-world writing prompts across 6 domains and 100 subdomains, each graded on five auto-generated, instance-specific criteria by an LLM or critic-model judge."
  },
  {
   "aliases": [],
   "canonical_id": "wsc273",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "wsc273",
   "models_covered": 0,
   "name": "WSC273",
   "reasons": [],
   "summary": "WSC273 scores pronoun-resolution accuracy on the first 273 items of the Winograd Schema Challenge, using language-model probability rather than fine-tuning."
  },
  {
   "aliases": [
    "Cross-lingual Choice of Plausible Alternatives"
   ],
   "canonical_id": "xcopa",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "xcopa",
   "models_covered": 0,
   "name": "XCOPA",
   "reasons": [],
   "summary": "XCOPA translates and re-annotates the English COPA causal-reasoning test into 11 typologically diverse languages, to measure zero-shot cross-lingual transfer of commonsense reasoning."
  },
  {
   "aliases": [],
   "canonical_id": "xiezhi",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "xiezhi",
   "models_covered": 0,
   "name": "Xiezhi",
   "reasons": [],
   "summary": "Xiezhi is a Chinese-and-English multiple-choice benchmark covering holistic knowledge across 516 academic disciplines in 13 subjects."
  },
  {
   "aliases": [],
   "canonical_id": "xl_docbench",
   "category": "long-context",
   "disposition": "unassessed",
   "id": "xl_docbench",
   "models_covered": 0,
   "name": "XL-DocBench",
   "reasons": [],
   "summary": "XL-DocBench tests evidence-grounded QA on extra-long professional documents, with page-level evidence, typed rules, and unanswerable cases."
  },
  {
   "aliases": [
    "XLSum",
    "XL-Sum",
    "csebuetnlp/xlsum"
   ],
   "canonical_id": "xlsum",
   "category": "generation",
   "disposition": "unassessed",
   "id": "xlsum",
   "models_covered": 0,
   "name": "XL-Sum",
   "reasons": [],
   "summary": "BBC article-summary pairs across 45 language configs; OpenCompass concatenates validation splits and scores ROUGE."
  },
  {
   "aliases": [
    "Cross-lingual NLI",
    "facebook/xnli"
   ],
   "canonical_id": "xnli",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "xnli",
   "models_covered": 0,
   "name": "XNLI (Cross-lingual Natural Language Inference)",
   "reasons": [],
   "summary": "Cross-lingual three-way NLI in 15 languages: given a premise and hypothesis, choose entailment, contradiction, or neutral."
  },
  {
   "aliases": [
    "xnli-eu",
    "HiTZ/xnli-eu",
    "XNLIeu"
   ],
   "canonical_id": "xnli_eu",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "xnli_eu",
   "models_covered": 0,
   "name": "XNLIeu",
   "reasons": [],
   "summary": "Basque XNLI: postedited and machine-translated English XNLI plus a 621-item native Basque test set, scored as three-way NLI accuracy."
  },
  {
   "aliases": [],
   "canonical_id": "xquad",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "xquad",
   "models_covered": 0,
   "name": "XQuAD",
   "reasons": [],
   "summary": "XQuAD is a cross-lingual extractive QA benchmark of 1,190 SQuAD v1.1 question-answer pairs professionally translated into 11 languages."
  },
  {
   "aliases": [],
   "canonical_id": "xstest",
   "category": "safety",
   "disposition": "unassessed",
   "id": "xstest",
   "models_covered": 0,
   "name": "XSTest",
   "reasons": [],
   "summary": "450 prompts, 250 safe and 200 minimally-edited unsafe contrasts across 10 categories, testing whether a model over-refuses safe requests that merely resemble unsafe ones."
  },
  {
   "aliases": [],
   "canonical_id": "xstorycloze",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "xstorycloze",
   "models_covered": 0,
   "name": "XStoryCloze",
   "reasons": [],
   "summary": "XStoryCloze translates the English StoryCloze validation set into 10 languages, testing commonsense selection of a story's correct ending."
  },
  {
   "aliases": [],
   "canonical_id": "xsum",
   "category": "generation",
   "disposition": "unassessed",
   "id": "xsum",
   "models_covered": 0,
   "name": "XSum",
   "reasons": [],
   "summary": "XSum is an English single-document summarization benchmark of 226,711 BBC articles paired with a one-sentence summary each."
  },
  {
   "aliases": [],
   "canonical_id": "xwinograd",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "xwinograd",
   "models_covered": 0,
   "name": "XWinograd",
   "reasons": [],
   "summary": "XWinograd is a multilingual Winograd Schema Challenge covering English, French, Japanese, Portuguese, Russian and Chinese."
  },
  {
   "aliases": [],
   "canonical_id": "yes_no_black_white",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "yes_no_black_white",
   "models_covered": 0,
   "name": "yes_no_black_white",
   "reasons": [],
   "summary": "yes_no_black_white is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "zebralogic",
   "category": "knowledge",
   "disposition": "unassessed",
   "id": "zebralogic",
   "models_covered": 0,
   "name": "ZebraLogic",
   "reasons": [],
   "summary": "ZebraLogic is a benchmark task documented by its cited evaluation harness."
  },
  {
   "aliases": [],
   "canonical_id": "zerobench",
   "category": "multimodal",
   "disposition": "unassessed",
   "id": "zerobench",
   "models_covered": 1,
   "name": "ZeroBench",
   "reasons": [],
   "summary": "100 hand-made visual reasoning questions filtered so no evaluated frontier model answered any correctly at release, plus 334 subquestions to track partial progress."
  },
  {
   "aliases": [
    "ZhoBLiMP: Chinese linguistic minimal pairs"
   ],
   "canonical_id": "zhoblimp",
   "category": "reasoning",
   "disposition": "unassessed",
   "id": "zhoblimp",
   "models_covered": 0,
   "name": "ZhoBLiMP",
   "reasons": [],
   "summary": "About 35,000 Chinese minimal pairs across 118 paradigms and 15 linguistic phenomena test language-model grammatical knowledge."
  }
 ],
 "build": {
  "built_at": "2026-09-09T16:56:50+00:00",
  "commit": "0a599558854c0e238c03a0f0d725239cb28f9d11",
  "eligibility_as_of": "2026-09-09"
 },
 "counts": {
  "active": 7,
  "alias": 1,
  "unassessed": 1091,
  "unverified": 7
 },
 "eligibility_as_of": "2026-09-09"
}