{"tasksets": [{"n_runnable": 6, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "aider_polyglot", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/aider_polyglot", "n_single_container": 225, "n_tasks": 225, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": ["cpp", "go", "java", "javascript", "python", "rust"], "difficulties": {"medium": 225}, "categories": ["coding-exercises"]}, {"n_runnable": 3, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "bigcodebench_hard_complete", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/bigcodebench_hard_complete", "n_single_container": 145, "n_tasks": 145, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 145}, "categories": ["python_programming"]}, {"n_runnable": 3, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "bird-bench", "domain": "data-sql", "task_kind": null, "environment": null, "grading": null, "path": "datasets/bird-bench", "n_single_container": 1534, "n_tasks": 1534, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 860, "hard": 231, "medium": 443}, "categories": ["database"]}, {"n_runnable": 3, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "crustbench", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/crustbench", "n_single_container": 100, "n_tasks": 100, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 65, "hard": 16, "medium": 19}, "categories": ["programming"]}, {"n_runnable": 3, "name": "DABstep (Data Agent Benchmark for Multi-step Reasoning)", "url_data": "https://huggingface.co/datasets/adyen/DABstep", "catalog_id": "dabstep", "owner_type": "company", "cited_by": [], "taskset": "dabstep", "domain": "data-sql", "task_kind": "multi-step payments data analysis questions", "environment": "container", "grading": "exact-match", "path": "datasets/dabstep", "n_single_container": 450, "n_tasks": 450, "domain_raw": "data-analysis", "owner_org": "Adyen", "url_repo": null, "url_paper": null, "license": "cc-by-4.0", "languages": [], "difficulties": {"easy": 72, "hard": 378}, "categories": ["data-analysis"]}, {"n_runnable": 3, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "humanevalfix", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/humanevalfix", "n_single_container": 164, "n_tasks": 164, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 77, "medium": 87}, "categories": ["debugging"]}, {"n_runnable": 3, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "quixbugs", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/quixbugs", "n_single_container": 80, "n_tasks": 80, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": ["java", "python"], "difficulties": {"medium": 80}, "categories": ["debugging"]}, {"n_runnable": 3, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "spider2-dbt", "domain": "data-sql", "task_kind": null, "environment": null, "grading": null, "path": "datasets/spider2-dbt", "n_single_container": 64, "n_tasks": 64, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 64}, "categories": ["programming"]}, {"n_runnable": 3, "name": "SWE-bench Verified", "url_data": "https://huggingface.co/datasets/princeton-nlp/SWE-bench_Verified", "catalog_id": "swe-bench-verified", "owner_type": "community", "cited_by": ["Claude Sonnet 5", "GLM-5", "Gemini 3.1 Pro card", "Kimi K2.6", "MiniMax M3", "Mistral Medium 3.5", "Nemotron 3 Ultra", "Nemotron 3.5 Lightning", "everyone; Epoch AI hub"], "taskset": "swebench-verified", "domain": "swe", "task_kind": "resolve real GitHub issues in Python repos", "environment": "container", "grading": "tests", "path": "datasets/swebench-verified", "n_single_container": 500, "n_tasks": 500, "domain_raw": "swe", "owner_org": "OpenAI & SWE-bench team (Princeton/Stanford)", "url_repo": "https://github.com/SWE-bench/SWE-bench", "url_paper": "https://arxiv.org/abs/2310.06770", "license": "MIT", "languages": [], "difficulties": {">4 hours": 3, "1-4 hours": 42, "<15 min fix": 194, "15 min - 1 hour": 261}, "categories": ["debugging"]}, {"n_runnable": 3, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "usaco", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/usaco", "n_single_container": 304, "n_tasks": 304, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 304}, "categories": ["python_programming"]}, {"n_runnable": 2, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "evoeval", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/evoeval", "n_single_container": 100, "n_tasks": 100, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 100}, "categories": ["python_programming"]}, {"n_runnable": 1, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "algotune", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/algotune", "n_single_container": 154, "n_tasks": 154, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 154}, "categories": ["algorithm"]}, {"n_runnable": 0, "name": "Artificial Analysis Long Context Reasoning (AA-LCR)", "url_data": "https://huggingface.co/datasets/ArtificialAnalysis/AA-LCR", "catalog_id": "aa-lcr", "owner_type": "company", "cited_by": ["Kimi K3", "Mistral Small 4", "Nemotron 3 Ultra", "Nemotron 3.5 Lightning"], "taskset": "aa-lcr", "domain": "reasoning-knowledge", "task_kind": "multi-document reasoning QA over ~100k-token inputs", "environment": "none", "grading": "llm-judge", "path": "datasets/aa-lcr", "n_single_container": 99, "n_tasks": 99, "domain_raw": "long-context", "owner_org": "Artificial Analysis", "url_repo": null, "url_paper": "https://artificialanalysis.ai/articles/announcing-aa-lcr", "license": "Apache-2.0", "languages": [], "difficulties": {"hard": 99}, "categories": ["long-context-reasoning"]}, {"n_runnable": 0, "name": "ABC-Bench", "url_data": null, "catalog_id": "abc-bench", "owner_type": "academic", "cited_by": [], "taskset": "abc-bench", "domain": "life-sciences", "task_kind": "agentic bio-capabilities (biosecurity)", "environment": "api-sim", "grading": "mixed", "path": "datasets/abc-bench", "n_single_container": 0, "n_tasks": 224, "domain_raw": "life-sciences", "owner_org": null, "url_repo": null, "url_paper": "https://arxiv.org/abs/2606.11150", "license": null, "languages": [], "difficulties": {"easy": 41, "medium": 44, "hard": 139}, "categories": ["Analytics", "Commerce", "Communication", "Content", "DevTools", "Entertainment", "Identity", "Infrastructure", "Other", "Specialized"]}, {"n_runnable": 0, "name": "ACE", "url_data": "https://huggingface.co/datasets/mercor/ACE", "catalog_id": "ace-mercor", "owner_type": "company", "cited_by": [], "taskset": "ace-bench", "domain": "enterprise-ops", "task_kind": "evaluation criteria for DIY/food/shopping/gaming tasks", "environment": "live-web", "grading": "rubric", "path": "datasets/ace-bench", "n_single_container": 973, "n_tasks": 973, "domain_raw": "consumer", "owner_org": "Mercor", "url_repo": null, "url_paper": null, "license": "cc-by-4.0", "languages": [], "difficulties": {"medium": 973}, "categories": ["tool-use"]}, {"n_runnable": 0, "name": "ACE", "url_data": "https://huggingface.co/datasets/mercor/ACE", "catalog_id": "ace-mercor", "owner_type": "company", "cited_by": [], "taskset": "acebench-normal", "domain": "enterprise-ops", "task_kind": "evaluation criteria for DIY/food/shopping/gaming tasks", "environment": "live-web", "grading": "rubric", "path": "datasets/acebench-normal", "n_single_container": 823, "n_tasks": 823, "domain_raw": "consumer", "owner_org": "Mercor", "url_repo": null, "url_paper": null, "license": "cc-by-4.0", "languages": [], "difficulties": {"medium": 823}, "categories": ["tool-use"]}, {"n_runnable": 0, "name": "ACE", "url_data": "https://huggingface.co/datasets/mercor/ACE", "catalog_id": "ace-mercor", "owner_type": "company", "cited_by": [], "taskset": "acebench-special", "domain": "enterprise-ops", "task_kind": "evaluation criteria for DIY/food/shopping/gaming tasks", "environment": "live-web", "grading": "rubric", "path": "datasets/acebench-special", "n_single_container": 150, "n_tasks": 150, "domain_raw": "consumer", "owner_org": "Mercor", "url_repo": null, "url_paper": null, "license": "cc-by-4.0", "languages": [], "difficulties": {"medium": 150}, "categories": ["tool-use"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "ade-bench", "domain": "data-sql", "task_kind": null, "environment": null, "grading": null, "path": "ade-bench", "n_single_container": 48, "n_tasks": 48, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 9, "medium": 27, "hard": 12}, "categories": ["data-engineering"]}, {"n_runnable": 0, "name": "AIME (American Invitational Mathematics Examination)", "url_data": "https://maa.org/maa-invitational-competitions/", "catalog_id": "aime", "owner_type": "community", "cited_by": ["GLM-5", "GLM-5.1", "GLM-5.2", "Kimi K2.6", "Mistral Medium 3.5", "Mistral Small 4"], "taskset": "aime", "domain": "science-math", "task_kind": "competition math, integer answers", "environment": "none", "grading": "exact-match", "path": "datasets/aime", "n_single_container": 60, "n_tasks": 60, "domain_raw": "math", "owner_org": "Mathematical Association of America", "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"unknown": 60}, "categories": ["reasoning"]}, {"n_runnable": 0, "name": "ARC-AGI-1", "url_data": "https://arcprize.org/arc-agi", "catalog_id": "arc-agi-1", "owner_type": "community", "cited_by": ["GPT-6 Astra"], "taskset": "arc_agi_1", "domain": "reasoning-knowledge", "task_kind": "abstract grid-puzzle induction", "environment": "none", "grading": "exact-match", "path": "datasets/arc_agi_1", "n_single_container": 400, "n_tasks": 400, "domain_raw": "reasoning", "owner_org": "ARC Prize Foundation", "url_repo": "https://github.com/fchollet/ARC-AGI", "url_paper": "https://arxiv.org/abs/1911.01547", "license": "Apache-2.0", "languages": [], "difficulties": {"hard": 400}, "categories": ["reasoning"]}, {"n_runnable": 0, "name": "ARC-AGI-2", "url_data": "https://www.kaggle.com/competitions/arc-prize-2026-arc-agi-2", "catalog_id": "arc-agi-2", "owner_type": "community", "cited_by": ["GPT-6 Astra"], "taskset": "arc_agi_2", "domain": "reasoning-knowledge", "task_kind": "abstract grid-puzzle induction", "environment": "none", "grading": "exact-match", "path": "datasets/arc_agi_2", "n_single_container": 167, "n_tasks": 167, "domain_raw": "reasoning", "owner_org": "ARC Prize Foundation", "url_repo": "https://github.com/arcprize/ARC-AGI-2", "url_paper": "https://arxiv.org/abs/2505.11831", "license": "Apache-2.0", "languages": [], "difficulties": {"hard": 167}, "categories": ["reasoning"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "autocodebench", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/autocodebench", "n_single_container": 200, "n_tasks": 200, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": ["cpp", "csharp", "dart", "elixir", "go", "java", "javascript", "julia", "kotlin", "perl", "php", "python", "r", "racket", "ruby", "rust", "scala", "shell", "swift", "typescript"], "difficulties": {"easy": 27, "hard": 127, "medium": 46}, "categories": ["coding"]}, {"n_runnable": 0, "name": "Berkeley Function Calling Leaderboard (BFCL)", "url_data": "https://huggingface.co/datasets/gorilla-llm/Berkeley-Function-Calling-Leaderboard", "catalog_id": "bfcl", "owner_type": "academic", "cited_by": [], "taskset": "bfcl", "domain": "tool-use", "task_kind": "function calling (single, parallel, multi-turn)", "environment": "api-sim", "grading": "state-check", "path": "datasets/bfcl", "n_single_container": 3641, "n_tasks": 3641, "domain_raw": "tool-use", "owner_org": "UC Berkeley (Gorilla)", "url_repo": "https://github.com/ShishirPatil/gorilla", "url_paper": null, "license": "apache-2.0", "languages": [], "difficulties": {"medium": 3641}, "categories": ["function_calling"]}, {"n_runnable": 0, "name": "Berkeley Function Calling Leaderboard (BFCL)", "url_data": "https://huggingface.co/datasets/gorilla-llm/Berkeley-Function-Calling-Leaderboard", "catalog_id": "bfcl", "owner_type": "academic", "cited_by": [], "taskset": "bfcl_parity", "domain": "tool-use", "task_kind": "function calling (single, parallel, multi-turn)", "environment": "api-sim", "grading": "state-check", "path": "datasets/bfcl_parity", "n_single_container": 123, "n_tasks": 123, "domain_raw": "tool-use", "owner_org": "UC Berkeley (Gorilla)", "url_repo": "https://github.com/ShishirPatil/gorilla", "url_paper": null, "license": "apache-2.0", "languages": [], "difficulties": {"medium": 123}, "categories": ["function_calling"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "bigcodebench_hard_instruct", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/bigcodebench_hard_instruct", "n_single_container": 145, "n_tasks": 145, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 145}, "categories": ["python_programming"]}, {"n_runnable": 0, "name": "BixBench", "url_data": "https://huggingface.co/datasets/futurehouse/BixBench", "catalog_id": "bixbench", "owner_type": "company", "cited_by": [], "taskset": "bixbench", "domain": "life-sciences", "task_kind": "computational biology analysis capsules in notebooks", "environment": "container", "grading": "mixed", "path": "datasets/bixbench", "n_single_container": 205, "n_tasks": 205, "domain_raw": "life-sciences", "owner_org": "FutureHouse", "url_repo": "https://github.com/Future-House/BixBench", "url_paper": "https://arxiv.org/abs/2503.00096", "license": "Apache-2.0", "languages": [], "difficulties": {"hard": 205}, "categories": ["computational_biology"]}, {"n_runnable": 0, "name": "BixBench", "url_data": "https://huggingface.co/datasets/futurehouse/BixBench", "catalog_id": "bixbench", "owner_type": "company", "cited_by": [], "taskset": "bixbench-cli", "domain": "life-sciences", "task_kind": "computational biology analysis capsules in notebooks", "environment": "container", "grading": "mixed", "path": "datasets/bixbench-cli", "n_single_container": 205, "n_tasks": 205, "domain_raw": "life-sciences", "owner_org": "FutureHouse", "url_repo": "https://github.com/Future-House/BixBench", "url_paper": "https://arxiv.org/abs/2503.00096", "license": "Apache-2.0", "languages": [], "difficulties": {"hard": 205}, "categories": ["computational_biology"]}, {"n_runnable": 0, "name": "CL-bench", "url_data": "https://huggingface.co/datasets/tencent/CL-bench", "catalog_id": "cl-bench", "owner_type": "company", "cited_by": ["MiniMax M3"], "taskset": "clbench", "domain": "reasoning-knowledge", "task_kind": "learn new knowledge from a given context and apply it", "environment": "none", "grading": "rubric", "path": "datasets/clbench", "n_single_container": 0, "n_tasks": 1899, "domain_raw": "long-context", "owner_org": "Tencent Hunyuan", "url_repo": "https://github.com/Tencent-Hunyuan/CL-bench", "url_paper": "https://arxiv.org/abs/2602.03587", "license": "other", "languages": [], "difficulties": {"hard": 1899}, "categories": ["context-learning"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "codepde", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/codepde", "n_single_container": 5, "n_tasks": 5, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 1, "medium": 2, "hard": 2}, "categories": ["scientific-computing"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "compilebench", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "hub-datasets/compilebench", "n_single_container": 15, "n_tasks": 15, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 2, "medium": 8, "hard": 5}, "categories": ["programming"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "cooperbench", "domain": "reasoning-knowledge", "task_kind": null, "environment": null, "grading": null, "path": "datasets/cooperbench", "n_single_container": 0, "n_tasks": 652, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"hard": 652}, "categories": ["multi-agent-cooperation"]}, {"n_runnable": 0, "name": "CRMArena", "url_data": "https://huggingface.co/datasets/Salesforce/CRMArena", "catalog_id": "crmarena", "owner_type": "company", "cited_by": [], "taskset": "crmarena", "domain": "customer-service", "task_kind": "query & act in a real Salesforce org (sandboxed) for service-agent tasks", "environment": "live-web", "grading": "exact-match", "path": "datasets/crmarena", "n_single_container": 1170, "n_tasks": 1170, "domain_raw": "CRM / customer service", "owner_org": "Salesforce AI Research", "url_repo": "https://github.com/SalesforceAIResearch/CRMArena", "url_paper": "https://arxiv.org/abs/2411.02305", "license": "CC-BY-NC-4.0", "languages": [], "difficulties": {"medium": 1170}, "categories": ["crm"]}, {"n_runnable": 0, "name": "CyberGym", "url_data": "https://huggingface.co/datasets/sunblaze-ucb/cybergym", "catalog_id": "cybergym", "owner_type": "academic", "cited_by": ["Claude Sonnet 5", "DeepSeek-V4-Flash-0731", "DeepSeek-V4-Flash-Vision-Exp", "DeepSeek-V4-Pro-0813", "DeepSeek-V4.1-Flash", "GLM-5", "GLM-5.1", "GLM-5.3"], "taskset": "cybergym", "domain": "security", "task_kind": "PoC generation to reproduce real vulnerabilities", "environment": "container", "grading": "state-check", "path": "datasets/cybergym", "n_single_container": 6028, "n_tasks": 6028, "domain_raw": "cyber", "owner_org": "UC Berkeley (Sunblaze / RDI)", "url_repo": "https://github.com/sunblaze-ucb/cybergym", "url_paper": "https://arxiv.org/abs/2506.02548", "license": "Apache-2.0", "languages": [], "difficulties": {"?": 6028}, "categories": []}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "dacode", "domain": "data-sql", "task_kind": null, "environment": null, "grading": null, "path": "datasets/dacode", "n_single_container": 479, "n_tasks": 479, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 94, "hard": 98, "medium": 287}, "categories": ["data-science"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "deepsynth", "domain": "web-research", "task_kind": null, "environment": null, "grading": null, "path": "datasets/deepsynth", "n_single_container": 40, "n_tasks": 40, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"difficult": 40}, "categories": ["information-synthesis"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "deveval", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/deveval", "n_single_container": 63, "n_tasks": 63, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": ["cpp", "java", "javascript", "python"], "difficulties": {"hard": 63}, "categories": ["software-development"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "devopsgym", "domain": "devops-sre", "task_kind": null, "environment": null, "grading": null, "path": "datasets/devopsgym", "n_single_container": 616, "n_tasks": 733, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"hard": 733}, "categories": ["build", "code-generation", "integration", "monitoring", "test-generation"]}, {"n_runnable": 0, "name": "DS-1000", "url_data": "https://huggingface.co/datasets/xlangai/DS-1000", "catalog_id": "ds-1000", "owner_type": "academic", "cited_by": [], "taskset": "ds1000", "domain": "data-sql", "task_kind": "data-science code generation", "environment": "container", "grading": "tests", "path": "datasets/ds1000", "n_single_container": 1000, "n_tasks": 1000, "domain_raw": "data-science", "owner_org": "XLANG Lab (HKU)", "url_repo": "https://github.com/xlang-ai/DS-1000", "url_paper": null, "license": "cc-by-sa-4.0", "languages": [], "difficulties": {"?": 1000}, "categories": []}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "featbench", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/featbench", "n_single_container": 156, "n_tasks": 156, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"hard": 156}, "categories": ["feature"]}, {"n_runnable": 0, "name": "FeatureBench", "url_data": "https://huggingface.co/datasets/LiberCoders/FeatureBench", "catalog_id": "featurebench", "owner_type": "academic", "cited_by": [], "taskset": "featurebench", "domain": "swe", "task_kind": "end-to-end feature implementation", "environment": "container", "grading": "tests", "path": "datasets/featurebench", "n_single_container": 156, "n_tasks": 200, "domain_raw": "swe", "owner_org": "LiberCoders", "url_repo": null, "url_paper": null, "license": "mit", "languages": [], "difficulties": {"medium": 166, "hard": 34}, "categories": ["feature"]}, {"n_runnable": 0, "name": "FeatureBench", "url_data": "https://huggingface.co/datasets/LiberCoders/FeatureBench", "catalog_id": "featurebench", "owner_type": "academic", "cited_by": [], "taskset": "featurebench-lite", "domain": "swe", "task_kind": "end-to-end feature implementation", "environment": "container", "grading": "tests", "path": "datasets/featurebench-lite", "n_single_container": 23, "n_tasks": 30, "domain_raw": "swe", "owner_org": "LiberCoders", "url_repo": null, "url_paper": null, "license": "mit", "languages": [], "difficulties": {"medium": 26, "hard": 4}, "categories": ["feature"]}, {"n_runnable": 0, "name": "FeatureBench", "url_data": "https://huggingface.co/datasets/LiberCoders/FeatureBench", "catalog_id": "featurebench", "owner_type": "academic", "cited_by": [], "taskset": "featurebench-lite-modal", "domain": "swe", "task_kind": "end-to-end feature implementation", "environment": "container", "grading": "tests", "path": "datasets/featurebench-lite-modal", "n_single_container": 23, "n_tasks": 30, "domain_raw": "swe", "owner_org": "LiberCoders", "url_repo": null, "url_paper": null, "license": "mit", "languages": [], "difficulties": {"medium": 26, "hard": 4}, "categories": ["feature"]}, {"n_runnable": 0, "name": "FeatureBench", "url_data": "https://huggingface.co/datasets/LiberCoders/FeatureBench", "catalog_id": "featurebench", "owner_type": "academic", "cited_by": [], "taskset": "featurebench-modal", "domain": "swe", "task_kind": "end-to-end feature implementation", "environment": "container", "grading": "tests", "path": "datasets/featurebench-modal", "n_single_container": 156, "n_tasks": 200, "domain_raw": "swe", "owner_org": "LiberCoders", "url_repo": null, "url_paper": null, "license": "mit", "languages": [], "difficulties": {"medium": 166, "hard": 34}, "categories": ["feature"]}, {"n_runnable": 0, "name": "Finance Agent Benchmark", "url_data": "https://huggingface.co/datasets/vals-ai/finance_agent_benchmark", "catalog_id": "vals-finance-agent", "owner_type": "company", "cited_by": ["Gemini 3.8 Flash", "Kimi K3", "Nemotron 3 Ultra"], "taskset": "financeagent", "domain": "finance", "task_kind": "financial research on SEC filings with tools", "environment": "live-web", "grading": "llm-judge", "path": "datasets/financeagent", "n_single_container": 50, "n_tasks": 50, "domain_raw": "finance", "owner_org": "Vals AI", "url_repo": "https://github.com/vals-ai/finance-agent", "url_paper": "https://arxiv.org/abs/2508.00828", "license": "CC-BY-4.0", "languages": [], "difficulties": {"?": 50}, "categories": ["financeQA"]}, {"n_runnable": 0, "name": "Finance Agent Benchmark", "url_data": "https://huggingface.co/datasets/vals-ai/finance_agent_benchmark", "catalog_id": "vals-finance-agent", "owner_type": "company", "cited_by": ["Gemini 3.8 Flash", "Kimi K3", "Nemotron 3 Ultra"], "taskset": "financeagent_terminal", "domain": "finance", "task_kind": "financial research on SEC filings with tools", "environment": "live-web", "grading": "llm-judge", "path": "datasets/financeagent_terminal", "n_single_container": 0, "n_tasks": 50, "domain_raw": "finance", "owner_org": "Vals AI", "url_repo": "https://github.com/vals-ai/finance-agent", "url_paper": "https://arxiv.org/abs/2508.00828", "license": "CC-BY-4.0", "languages": [], "difficulties": {"?": 50}, "categories": ["financeQA"]}, {"n_runnable": 0, "name": "Frontier-CS", "url_data": null, "catalog_id": "frontier-cs", "owner_type": "academic", "cited_by": ["inspect_evals"], "taskset": "frontier-cs", "domain": "swe", "task_kind": "open-ended algorithmic + research CS problems (GPU kernels, symbolic regression)", "environment": "container", "grading": "tests", "path": "hub-datasets/frontier-cs", "n_single_container": 0, "n_tasks": 172, "domain_raw": "software-engineering", "owner_org": null, "url_repo": "https://github.com/FrontierCS/Frontier-CS", "url_paper": "https://arxiv.org/abs/2512.15699", "license": "MIT", "languages": [], "difficulties": {"hard": 172}, "categories": ["competitive-programming"]}, {"n_runnable": 0, "name": "GAIA", "url_data": "https://huggingface.co/datasets/gaia-benchmark/GAIA", "catalog_id": "gaia", "owner_type": "lab", "cited_by": ["HAL", "Steel leaderboard", "inspect_evals"], "taskset": "gaia", "domain": "tool-use", "task_kind": "web + file + tool multi-step questions with short answers", "environment": "live-web", "grading": "exact-match", "path": "datasets/gaia", "n_single_container": 165, "n_tasks": 165, "domain_raw": "general assistant", "owner_org": "Meta + Hugging Face", "url_repo": null, "url_paper": "https://arxiv.org/abs/2311.12983", "license": null, "languages": [], "difficulties": {"medium": 165}, "categories": ["reasoning"]}, {"n_runnable": 0, "name": "Gaia2 (on ARE)", "url_data": "https://huggingface.co/datasets/meta-agents-research-environments/gaia2", "catalog_id": "gaia2", "owner_type": "lab", "cited_by": ["HF leaderboard space"], "taskset": "gaia2", "domain": "customer-service", "task_kind": "asynchronous, time-sensitive, multi-agent scenarios with oracle events", "environment": "api-sim", "grading": "state-check", "path": "datasets/gaia2", "n_single_container": 800, "n_tasks": 800, "domain_raw": "personal assistant on simulated smartphone (email, calendar, contacts, files, messaging)", "owner_org": "Meta", "url_repo": "https://github.com/facebookresearch/meta-agents-research-environments", "url_paper": "https://arxiv.org/abs/2602.11964", "license": "CC-BY-4.0", "languages": [], "difficulties": {"easy": 182, "hard": 270, "medium": 348}, "categories": ["adaptability", "ambiguity", "execution", "search", "time"]}, {"n_runnable": 0, "name": "Gaia2 (on ARE)", "url_data": "https://huggingface.co/datasets/meta-agents-research-environments/gaia2", "catalog_id": "gaia2", "owner_type": "lab", "cited_by": ["HF leaderboard space"], "taskset": "gaia2-cli", "domain": "customer-service", "task_kind": "asynchronous, time-sensitive, multi-agent scenarios with oracle events", "environment": "api-sim", "grading": "state-check", "path": "datasets/gaia2-cli", "n_single_container": 0, "n_tasks": 800, "domain_raw": "personal assistant on simulated smartphone (email, calendar, contacts, files, messaging)", "owner_org": "Meta", "url_repo": "https://github.com/facebookresearch/meta-agents-research-environments", "url_paper": "https://arxiv.org/abs/2602.11964", "license": "CC-BY-4.0", "languages": [], "difficulties": {"easy": 182, "hard": 270, "medium": 348}, "categories": ["adaptability", "ambiguity", "execution", "search", "time"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "gdb", "domain": "office-docs", "task_kind": null, "environment": null, "grading": null, "path": "datasets/gdb", "n_single_container": 78, "n_tasks": 78, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 6, "medium": 22, "hard": 50}, "categories": ["design"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "gdb-hub", "domain": "office-docs", "task_kind": null, "environment": null, "grading": null, "path": "hub-datasets/gdb", "n_single_container": 33786, "n_tasks": 33786, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 4921, "medium": 17307, "hard": 11558}, "categories": ["design"]}, {"n_runnable": 0, "name": "GPQA Diamond", "url_data": "https://huggingface.co/datasets/Idavidrein/gpqa", "catalog_id": "gpqa-diamond", "owner_type": "academic", "cited_by": ["DeepSeek-V4.1-Flash", "GLM-5", "GLM-5.1", "GLM-5.2", "GPT-5.6 Sol", "GPT-6 Astra", "Kimi K2.6", "Kimi K3", "Nemotron 3 Ultra", "Nemotron 3.5 Lightning", "Qwen3.8-27B", "Qwen3.8-Flash-Next", "Qwen3.8-Max"], "taskset": "gpqa-diamond", "domain": "science-math", "task_kind": "graduate-level science multiple choice", "environment": "none", "grading": "exact-match", "path": "datasets/gpqa-diamond", "n_single_container": 198, "n_tasks": 198, "domain_raw": "science", "owner_org": "NYU / Cohere / Anthropic (Rein et al.)", "url_repo": "https://github.com/idavidrein/gpqa", "url_paper": "https://arxiv.org/abs/2311.12022", "license": "MIT", "languages": [], "difficulties": {"difficult": 198}, "categories": ["reasoning"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "gso", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/gso", "n_single_container": 102, "n_tasks": 102, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"hard": 102}, "categories": ["performance_optimization"]}, {"n_runnable": 0, "name": "Humanity's Last Exam (HLE)", "url_data": "https://huggingface.co/datasets/cais/hle", "catalog_id": "humanitys-last-exam", "owner_type": "community", "cited_by": ["Claude Fable 5.1", "Claude Opus 5.5", "Claude Sonnet 5", "DeepSeek-V4-Pro-0813", "DeepSeek-V4.1-Flash", "GLM-5", "GLM-5.1", "GLM-5.2", "GLM-5.3", "GLM-5.3-Flash", "GPT-6 Astra", "Kimi K2.6", "Kimi K3", "Nemotron 3 Ultra", "Nemotron 3.5 Lightning", "Qwen3.8-27B", "Qwen3.8-Flash-Next", "Qwen3.8-Max"], "taskset": "hle", "domain": "reasoning-knowledge", "task_kind": "expert-level closed-ended questions (multimodal)", "environment": "none", "grading": "llm-judge", "path": "datasets/hle", "n_single_container": 2500, "n_tasks": 2500, "domain_raw": "knowledge", "owner_org": "Center for AI Safety & Scale AI", "url_repo": "https://github.com/centerforaisafety/hle", "url_paper": "https://arxiv.org/abs/2501.14249", "license": "MIT", "languages": [], "difficulties": {"?": 2500}, "categories": []}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "ineqmath", "domain": "science-math", "task_kind": null, "environment": null, "grading": null, "path": "datasets/ineqmath", "n_single_container": 100, "n_tasks": 100, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"difficult": 100}, "categories": ["math reasoning"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "kramabench", "domain": "data-sql", "task_kind": null, "environment": null, "grading": null, "path": "datasets/kramabench", "n_single_container": 104, "n_tasks": 104, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 42, "hard": 62}, "categories": ["data-science"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "kumo", "domain": "reasoning-knowledge", "task_kind": null, "environment": null, "grading": null, "path": "datasets/kumo", "n_single_container": 0, "n_tasks": 5300, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 5300}, "categories": ["reasoning"]}, {"n_runnable": 0, "name": "LAB-Bench", "url_data": "https://huggingface.co/datasets/futurehouse/lab-bench", "catalog_id": "lab-bench", "owner_type": "company", "cited_by": [], "taskset": "labbench", "domain": "life-sciences", "task_kind": "biology research skills (LitQA2, DbQA, Cloning...)", "environment": "none", "grading": "exact-match", "path": "datasets/labbench", "n_single_container": 181, "n_tasks": 181, "domain_raw": "biology", "owner_org": "FutureHouse", "url_repo": null, "url_paper": "https://arxiv.org/abs/2407.10362", "license": "cc-by-sa-4.0", "languages": [], "difficulties": {"medium": 181}, "categories": ["multiple-choice"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "lawbench", "domain": "legal", "task_kind": null, "environment": null, "grading": null, "path": "datasets/lawbench", "n_single_container": 1000, "n_tasks": 1000, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 1000}, "categories": ["legal NLP benchmark"]}, {"n_runnable": 0, "name": "LiveCodeBench", "url_data": "https://huggingface.co/datasets/livecodebench/code_generation_lite", "catalog_id": "livecodebench", "owner_type": "academic", "cited_by": ["Kimi K2.6", "Mistral Small 4", "Nemotron 3 Ultra", "Qwen3.8-27B", "Qwen3.8-Flash-Next"], "taskset": "livecodebench", "domain": "swe", "task_kind": "contamination-free competitive programming problems", "environment": "none", "grading": "tests", "path": "datasets/livecodebench", "n_single_container": 100, "n_tasks": 100, "domain_raw": "coding", "owner_org": "LiveCodeBench team (UC Berkeley/MIT/Cornell)", "url_repo": "https://github.com/LiveCodeBench/LiveCodeBench", "url_paper": "https://arxiv.org/abs/2403.07974", "license": "MIT", "languages": [], "difficulties": {"?": 100}, "categories": []}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "llmsr-bench", "domain": "science-math", "task_kind": null, "environment": null, "grading": null, "path": "datasets/llmsr-bench", "n_single_container": 240, "n_tasks": 240, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 129, "hard": 111}, "categories": ["scientific-equation-discovery"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "locomo", "domain": "reasoning-knowledge", "task_kind": null, "environment": null, "grading": null, "path": "datasets/locomo", "n_single_container": 10, "n_tasks": 10, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"hard": 10}, "categories": ["memory-qa"]}, {"n_runnable": 0, "name": "MedAgentBench", "url_data": null, "catalog_id": "medagentbench", "owner_type": "academic", "cited_by": [], "taskset": "medagentbench", "domain": "healthcare", "task_kind": "FHIR EHR tasks in a virtual EHR environment", "environment": "container", "grading": "state-check", "path": "medagentbench", "n_single_container": 300, "n_tasks": 300, "domain_raw": "healthcare", "owner_org": "Stanford", "url_repo": "https://github.com/stanfordmlgroup/MedAgentBench", "url_paper": null, "license": "MIT", "languages": [], "difficulties": {"hard": 300}, "categories": ["medical-agent"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "ml_dev_bench", "domain": "ml-research", "task_kind": null, "environment": null, "grading": null, "path": "datasets/ml_dev_bench", "n_single_container": 0, "n_tasks": 33, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"hard": 21, "medium": 12}, "categories": ["data-processing", "debugging", "machine-learning", "system-integration"]}, {"n_runnable": 0, "name": "MLGym-Bench", "url_data": null, "catalog_id": "mlgym", "owner_type": "lab", "cited_by": [], "taskset": "mlgym-bench", "domain": "ml-research", "task_kind": "open-ended ML research tasks in a Gym env (CV, NLP, RL, game theory)", "environment": "container", "grading": "tests", "path": "datasets/mlgym-bench", "n_single_container": 11, "n_tasks": 11, "domain_raw": "AI research", "owner_org": "Meta", "url_repo": "https://github.com/facebookresearch/MLGym", "url_paper": "https://arxiv.org/abs/2502.14499", "license": "NOASSERTION", "languages": [], "difficulties": {"medium": 11}, "categories": ["machine-learning"]}, {"n_runnable": 0, "name": "MLGym-Bench", "url_data": null, "catalog_id": "mlgym", "owner_type": "lab", "cited_by": [], "taskset": "mlgym-bench-hub", "domain": "ml-research", "task_kind": "open-ended ML research tasks in a Gym env (CV, NLP, RL, game theory)", "environment": "container", "grading": "tests", "path": "hub-datasets/mlgym-bench", "n_single_container": 12, "n_tasks": 12, "domain_raw": "AI research", "owner_org": "Meta", "url_repo": "https://github.com/facebookresearch/MLGym", "url_paper": "https://arxiv.org/abs/2502.14499", "license": "NOASSERTION", "languages": [], "difficulties": {"easy": 12}, "categories": ["machine-learning"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "mmau", "domain": "multimodal", "task_kind": null, "environment": null, "grading": null, "path": "datasets/mmau", "n_single_container": 1000, "n_tasks": 1000, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 224, "medium": 540, "hard": 236}, "categories": ["audio"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "mmmlu", "domain": "science-math", "task_kind": null, "environment": null, "grading": null, "path": "datasets/mmmlu", "n_single_container": 150, "n_tasks": 150, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": ["AR-XY", "BN-BD", "DE-DE", "EN-US", "ES-LA", "FR-FR", "HI-IN", "ID-ID", "IT-IT", "JA-JP", "KO-KR", "PT-BR", "SW-KE", "YO-NG", "ZH-CN"], "difficulties": {"?": 150}, "categories": ["humanities", "other", "social_sciences", "stem"]}, {"n_runnable": 0, "name": "Multi-SWE-bench", "url_data": "https://huggingface.co/datasets/ByteDance-Seed/Multi-SWE-bench", "catalog_id": "multi-swe-bench", "owner_type": "lab", "cited_by": ["MiniMax M2.7"], "taskset": "multi-swe-bench", "domain": "swe", "task_kind": "multilingual GitHub issue resolution", "environment": "container", "grading": "tests", "path": "datasets/multi-swe-bench", "n_single_container": 1601, "n_tasks": 1601, "domain_raw": "swe", "owner_org": "ByteDance Seed", "url_repo": "https://github.com/multi-swe-bench/multi-swe-bench", "url_paper": "https://arxiv.org/abs/2504.02605", "license": "Apache-2.0", "languages": [], "difficulties": {"easy": 125, "medium": 773, "hard": 703}, "categories": ["software-development"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "omnimath", "domain": "science-math", "task_kind": null, "environment": null, "grading": null, "path": "datasets/omnimath", "n_single_container": 4428, "n_tasks": 4428, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"hard": 4428}, "categories": ["math"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "osworld-external-credentials", "domain": "other", "task_kind": null, "environment": null, "grading": null, "path": "optional-tasks/osworld-external-credentials", "n_single_container": 0, "n_tasks": 8, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"?": 8}, "categories": []}, {"n_runnable": 0, "name": "OSWorld-Verified", "url_data": "https://os-world.github.io", "catalog_id": "osworld-verified", "owner_type": "academic", "cited_by": ["Claude Sonnet 5", "Kimi K2.6", "Kimi K3", "MiniMax M3", "Qwen3.8-27B"], "taskset": "osworld-verified", "domain": "computer-use", "task_kind": "open-ended desktop/web tasks in real OS", "environment": "vm", "grading": "state-check", "path": "osworld-verified", "n_single_container": 0, "n_tasks": 361, "domain_raw": "computer-use", "owner_org": "XLANG Lab (HKU)", "url_repo": "https://github.com/xlang-ai/OSWorld", "url_paper": "https://arxiv.org/abs/2404.07972", "license": "Apache-2.0", "languages": [], "difficulties": {"?": 361}, "categories": []}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "pixiu", "domain": "finance", "task_kind": null, "environment": null, "grading": null, "path": "datasets/pixiu", "n_single_container": 435, "n_tasks": 435, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"?": 435}, "categories": []}, {"n_runnable": 0, "name": "ProgramBench", "url_data": "https://huggingface.co/datasets/programbench/ProgramBench-Tests", "catalog_id": "programbench", "owner_type": "lab", "cited_by": ["Claude Opus 5.5", "Claude Sonnet 5", "DeepSeek-V4.1-Flash", "GLM-5.2", "GLM-5.3", "Kimi K2.7-Code", "Kimi K3"], "taskset": "programbench", "domain": "swe", "task_kind": "rebuild a program from scratch to match a reference executable", "environment": "container", "grading": "tests", "path": "hub-datasets/programbench", "n_single_container": 200, "n_tasks": 200, "domain_raw": "swe", "owner_org": "Meta FAIR (facebookresearch)", "url_repo": "https://github.com/facebookresearch/ProgramBench", "url_paper": "https://arxiv.org/abs/2605.03546", "license": "MIT", "languages": ["c", "cpp", "go", "hs", "java", "rs"], "difficulties": {"medium": 120, "hard": 18, "easy": 27, "unknown": 35}, "categories": []}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "qcircuitbench", "domain": "science-math", "task_kind": null, "environment": null, "grading": null, "path": "datasets/qcircuitbench", "n_single_container": 28, "n_tasks": 28, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 8, "hard": 8, "medium": 12}, "categories": ["quantum"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "reasoning-gym", "domain": "reasoning-knowledge", "task_kind": null, "environment": null, "grading": null, "path": "datasets/reasoning-gym", "n_single_container": 576, "n_tasks": 576, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"?": 576}, "categories": []}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "refav", "domain": "engineering-industrial", "task_kind": null, "environment": null, "grading": null, "path": "hub-datasets/refav", "n_single_container": 1500, "n_tasks": 1500, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 1500}, "categories": ["scenario_mining"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "replicationbench", "domain": "web-research", "task_kind": null, "environment": null, "grading": null, "path": "datasets/replicationbench", "n_single_container": 90, "n_tasks": 90, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 37, "hard": 12, "medium": 41}, "categories": ["research"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "research-code-bench", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/research-code-bench", "n_single_container": 212, "n_tasks": 212, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"difficult": 212}, "categories": ["code-generation"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "rexbench", "domain": "ml-research", "task_kind": null, "environment": null, "grading": null, "path": "datasets/rexbench", "n_single_container": 2, "n_tasks": 2, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 2}, "categories": ["machine-learning"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "satbench", "domain": "reasoning-knowledge", "task_kind": null, "environment": null, "grading": null, "path": "datasets/satbench", "n_single_container": 2100, "n_tasks": 2100, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 2100}, "categories": ["reasoning"]}, {"n_runnable": 0, "name": "SciCode", "url_data": "https://huggingface.co/datasets/SciCode1/SciCode", "catalog_id": "scicode", "owner_type": "academic", "cited_by": ["Kimi K2.6", "Kimi K3", "Nemotron 3 Ultra", "Nemotron 3.5 Lightning"], "taskset": "scicode", "domain": "science-math", "task_kind": "scientific research coding subproblems", "environment": "container", "grading": "tests", "path": "datasets/scicode", "n_single_container": 0, "n_tasks": 80, "domain_raw": "science", "owner_org": "SciCode team (UIUC et al.)", "url_repo": "https://github.com/scicode-bench/SciCode", "url_paper": "https://arxiv.org/abs/2407.13168", "license": "Apache-2.0", "languages": [], "difficulties": {"hard": 80}, "categories": ["scientific_computing"]}, {"n_runnable": 0, "name": "ScienceAgentBench", "url_data": "https://huggingface.co/datasets/osunlp/ScienceAgentBench", "catalog_id": "scienceagentbench", "owner_type": "academic", "cited_by": [], "taskset": "scienceagentbench", "domain": "science-math", "task_kind": "data-driven scientific discovery programs", "environment": "container", "grading": "tests", "path": "hub-datasets/scienceagentbench", "n_single_container": 102, "n_tasks": 102, "domain_raw": "science", "owner_org": "OSU NLP", "url_repo": "https://github.com/OSU-NLP-Group/ScienceAgentBench", "url_paper": "https://arxiv.org/abs/2410.05080", "license": "cc-by-4.0", "languages": [], "difficulties": {"hard": 23, "medium": 79}, "categories": ["scientific_computing"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "seal0", "domain": "reasoning-knowledge", "task_kind": null, "environment": null, "grading": null, "path": "datasets/seal0", "n_single_container": 111, "n_tasks": 111, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"difficult": 111}, "categories": ["reasoning"]}, {"n_runnable": 0, "name": "SimpleQA", "url_data": "https://www.kaggle.com/benchmarks/openai/simpleqa", "catalog_id": "simpleqa", "owner_type": "lab", "cited_by": [], "taskset": "simpleqa", "domain": "reasoning-knowledge", "task_kind": "A benchmark from OpenAI designed to evaluate short-form factuality in large language models.", "environment": "none", "grading": "mixed", "path": "datasets/simpleqa", "n_single_container": 4326, "n_tasks": 4326, "domain_raw": "knowledge", "owner_org": "OpenAI", "url_repo": null, "url_paper": "https://arxiv.org/abs/2411.04368", "license": null, "languages": [], "difficulties": {"medium": 4326}, "categories": ["factuality"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "sldbench", "domain": "science-math", "task_kind": null, "environment": null, "grading": null, "path": "datasets/sldbench", "n_single_container": 8, "n_tasks": 8, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"medium": 8}, "categories": ["scientific_discovery"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "spreadsheetbench-verified", "domain": "office-docs", "task_kind": null, "environment": null, "grading": null, "path": "datasets/spreadsheetbench-verified", "n_single_container": 400, "n_tasks": 400, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"hard": 400}, "categories": ["spreadsheet-manipulation"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "strongreject", "domain": "ai-safety", "task_kind": null, "environment": null, "grading": null, "path": "datasets/strongreject", "n_single_container": 150, "n_tasks": 150, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"?": 150}, "categories": ["safety"]}, {"n_runnable": 0, "name": "SWE-Lancer", "url_data": null, "catalog_id": "swe-lancer", "owner_type": "lab", "cited_by": ["inspect_evals"], "taskset": "swe-lancer", "domain": "swe", "task_kind": "IC SWE patches + SWE-manager proposal selection, priced in USD", "environment": "container", "grading": "tests", "path": "datasets/swe-lancer", "n_single_container": 0, "n_tasks": 463, "domain_raw": "freelance software engineering (Expensify/Upwork)", "owner_org": "OpenAI", "url_repo": "https://github.com/openai/SWELancer-Benchmark", "url_paper": "https://arxiv.org/abs/2502.12115", "license": null, "languages": [], "difficulties": {"hard": 463}, "categories": ["debugging"]}, {"n_runnable": 0, "name": "SWE-bench Multilingual", "url_data": "https://huggingface.co/datasets/SWE-bench/SWE-bench_Multilingual", "catalog_id": "swe-bench-multilingual", "owner_type": "academic", "cited_by": ["Claude Opus 5.5", "Claude Sonnet 5", "GLM-5", "Kimi K2.6", "MiniMax M2.7", "Nemotron 3 Ultra", "Nemotron 3.5 Lightning", "Qwen3.8-Flash-Next"], "taskset": "swebench_multilingual", "domain": "swe", "task_kind": "GitHub issue resolution in 9 languages", "environment": "container", "grading": "tests", "path": "datasets/swebench_multilingual", "n_single_container": 300, "n_tasks": 300, "domain_raw": "swe", "owner_org": "SWE-bench team (Princeton/Stanford)", "url_repo": "https://github.com/SWE-bench/SWE-bench", "url_paper": "https://arxiv.org/abs/2504.21798", "license": "MIT", "languages": [], "difficulties": {"hard": 300}, "categories": ["debugging"]}, {"n_runnable": 0, "name": "SWE-Bench Pro", "url_data": "https://huggingface.co/datasets/ScaleAI/SWE-bench_Pro", "catalog_id": "swe-bench-pro", "owner_type": "company", "cited_by": ["Claude Opus 5.5", "Claude Sonnet 5", "GLM-5.1", "GLM-5.2", "GPT-5.6 Sol", "Kimi K2.6", "MiniMax M2.7", "MiniMax M3", "Qwen3.8-27B", "Qwen3.8-Flash-Next", "Qwen3.8-Max"], "taskset": "swebenchpro", "domain": "swe", "task_kind": "long-horizon repo issue resolution", "environment": "container", "grading": "tests", "path": "datasets/swebenchpro", "n_single_container": 731, "n_tasks": 731, "domain_raw": "swe", "owner_org": "Scale AI", "url_repo": "https://github.com/scaleapi/SWE-bench_Pro-os", "url_paper": "https://arxiv.org/abs/2509.16941", "license": "MIT", "languages": [], "difficulties": {"medium": 731}, "categories": ["debugging"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "swegym", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/swegym", "n_single_container": 2438, "n_tasks": 2438, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"hard": 2438}, "categories": ["debugging"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "swegym-lite", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/swegym-lite", "n_single_container": 230, "n_tasks": 230, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"hard": 230}, "categories": ["debugging"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "swesmith", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/swesmith", "n_single_container": 100, "n_tasks": 100, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"hard": 100}, "categories": ["debugging"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "swtbench-verified", "domain": "swe", "task_kind": null, "environment": null, "grading": null, "path": "datasets/swtbench-verified", "n_single_container": 433, "n_tasks": 433, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"?": 433}, "categories": ["test_generation"]}, {"n_runnable": 0, "name": "\u03c4\u00b3-bench", "url_data": null, "catalog_id": "tau3-bench", "owner_type": "company", "cited_by": ["GLM-5.1", "Kimi K3", "Mistral Medium 3.5", "Nemotron 3 Ultra", "Nemotron 3.5 Lightning"], "taskset": "tau3-bench", "domain": "customer-service", "task_kind": "tool-agent-user conversations (airline, retail, telecom, banking_knowledge)", "environment": "api-sim", "grading": "state-check", "path": "tau3-bench", "n_single_container": 0, "n_tasks": 375, "domain_raw": "customer-service", "owner_org": "Sierra", "url_repo": "https://github.com/sierra-research/tau2-bench", "url_paper": "https://arxiv.org/abs/2603.04370", "license": "MIT", "languages": [], "difficulties": {"medium": 375}, "categories": ["customer_service"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "textarena", "domain": "games", "task_kind": null, "environment": null, "grading": null, "path": "textarena", "n_single_container": 62, "n_tasks": 62, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 7, "medium": 35, "hard": 20}, "categories": ["games"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "theagentcompany", "domain": "enterprise-ops", "task_kind": null, "environment": null, "grading": null, "path": "theagentcompany", "n_single_container": 174, "n_tasks": 174, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"easy": 2, "medium": 73, "hard": 99}, "categories": ["admin", "bm", "ds", "finance", "hr", "ml", "pm", "qa", "research", "sde"]}, {"n_runnable": 0, "name": null, "url_data": null, "catalog_id": null, "owner_type": null, "cited_by": null, "taskset": "webgen-bench", "domain": "web-research", "task_kind": null, "environment": null, "grading": null, "path": "hub-datasets/webgen-bench", "n_single_container": 101, "n_tasks": 101, "domain_raw": null, "owner_org": null, "url_repo": null, "url_paper": null, "license": null, "languages": [], "difficulties": {"?": 101}, "categories": ["web_development"]}, {"n_runnable": 0, "name": "WideSearch", "url_data": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch", "catalog_id": "widesearch", "owner_type": "company", "cited_by": ["Kimi K2.6", "Qwen3.8-Max"], "taskset": "widesearch", "domain": "web-research", "task_kind": "broad info-seeking to fill structured tables", "environment": "live-web", "grading": "mixed", "path": "datasets/widesearch", "n_single_container": 200, "n_tasks": 200, "domain_raw": "web-browsing", "owner_org": "ByteDance Seed", "url_repo": "https://github.com/ByteDance-Seed/WideSearch", "url_paper": "https://arxiv.org/abs/2508.07999", "license": "MIT", "languages": [], "difficulties": {"hard": 200}, "categories": ["information-retrieval"]}], "domains": ["ai-safety", "computer-use", "customer-service", "data-sql", "devops-sre", "engineering-industrial", "enterprise-ops", "finance", "games", "healthcare", "legal", "life-sciences", "ml-research", "multimodal", "office-docs", "other", "reasoning-knowledge", "science-math", "security", "swe", "tool-use", "web-research"]}